{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# \n# model debugging\none of the debugging process is to debug the model after data debugging\n\n# \n\n# goal\nof this notebook is to build a simple [zero rule] baseline model \n\n# why\na model that has **no training and intelligence**, very **simple to implement**, can perform **quick prediction**, with **reproducible result**, will help to compare against future complex model for **debugging**. for instance, a new model score is lower than this model, then this new model needs tuning or try different model, that is one of the ways to identify your new **model performance**\n\n# how\nbuild labels from samples, predict by highest-count from labels\n\n# ","metadata":{}},{"cell_type":"code","source":"DEBUG = True","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.178251Z","iopub.execute_input":"2023-04-10T00:38:50.178684Z","iopub.status.idle":"2023-04-10T00:38:50.183832Z","shell.execute_reply.started":"2023-04-10T00:38:50.178633Z","shell.execute_reply":"2023-04-10T00:38:50.182466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"   import os\n   import math\n   import sys\n   from datetime import datetime\n   import psutil  \n   import numpy as np\n \n   def sys_stats():\n      pid = os.getpid()\n      ps = psutil.Process(pid)\n      memory_usage = ps.memory_info()[0] / 2. ** 30\n      log.info(f'{datetime.now()}  memory usage GB:' + str(np.round(memory_usage, 2)))","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.185668Z","iopub.execute_input":"2023-04-10T00:38:50.186256Z","iopub.status.idle":"2023-04-10T00:38:50.197545Z","shell.execute_reply.started":"2023-04-10T00:38:50.186219Z","shell.execute_reply":"2023-04-10T00:38:50.196566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import logging\nimport sys\n\n#log = logging.getLogger(\"gpt\")\nlog = logging.getLogger('')\nlog.setLevel(logging.DEBUG)\ncv = logging.StreamHandler(sys.stdout)\nif (log.hasHandlers()):\n    log.handlers.clear()\nlog.addHandler(cv)\nlog.propagate = False","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.198870Z","iopub.execute_input":"2023-04-10T00:38:50.199565Z","iopub.status.idle":"2023-04-10T00:38:50.211452Z","shell.execute_reply.started":"2023-04-10T00:38:50.199526Z","shell.execute_reply":"2023-04-10T00:38:50.210435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport gc","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.214002Z","iopub.execute_input":"2023-04-10T00:38:50.214673Z","iopub.status.idle":"2023-04-10T00:38:50.223062Z","shell.execute_reply.started":"2023-04-10T00:38:50.214624Z","shell.execute_reply":"2023-04-10T00:38:50.221531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_path = '/kaggle/input/predict-student-performance-from-game-play'\n","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.224609Z","iopub.execute_input":"2023-04-10T00:38:50.225305Z","iopub.status.idle":"2023-04-10T00:38:50.233442Z","shell.execute_reply.started":"2023-04-10T00:38:50.225266Z","shell.execute_reply":"2023-04-10T00:38:50.232458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_header = pd.read_csv(f'{base_path}/train_labels.csv', index_col=0, nrows=0).columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.235103Z","iopub.execute_input":"2023-04-10T00:38:50.235784Z","iopub.status.idle":"2023-04-10T00:38:50.252584Z","shell.execute_reply.started":"2023-04-10T00:38:50.235745Z","shell.execute_reply":"2023-04-10T00:38:50.251264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_data = pd.read_csv(f'{base_path}/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.254192Z","iopub.execute_input":"2023-04-10T00:38:50.254855Z","iopub.status.idle":"2023-04-10T00:38:50.580564Z","shell.execute_reply.started":"2023-04-10T00:38:50.254801Z","shell.execute_reply":"2023-04-10T00:38:50.579617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_data[['session_id','question_id']] = train_data['session_id'].str.split('_',expand=True)\ntrain_data = train_data.rename(columns={'correct': 'label'})","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.581978Z","iopub.execute_input":"2023-04-10T00:38:50.582531Z","iopub.status.idle":"2023-04-10T00:38:50.595601Z","shell.execute_reply.started":"2023-04-10T00:38:50.582495Z","shell.execute_reply":"2023-04-10T00:38:50.594517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.597038Z","iopub.execute_input":"2023-04-10T00:38:50.597752Z","iopub.status.idle":"2023-04-10T00:38:50.609194Z","shell.execute_reply.started":"2023-04-10T00:38:50.597712Z","shell.execute_reply":"2023-04-10T00:38:50.607895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_data_splitted, test_data_splitted = train_test_split(train_data, test_size=0.2, random_state=11)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.613107Z","iopub.execute_input":"2023-04-10T00:38:50.613463Z","iopub.status.idle":"2023-04-10T00:38:50.695586Z","shell.execute_reply.started":"2023-04-10T00:38:50.613427Z","shell.execute_reply":"2023-04-10T00:38:50.694443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### monitoring the memory usage\ngc.collect()\nsys_stats()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.696988Z","iopub.execute_input":"2023-04-10T00:38:50.697570Z","iopub.status.idle":"2023-04-10T00:38:50.826973Z","shell.execute_reply.started":"2023-04-10T00:38:50.697533Z","shell.execute_reply":"2023-04-10T00:38:50.825920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_splitted['label'].values","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.828459Z","iopub.execute_input":"2023-04-10T00:38:50.829613Z","iopub.status.idle":"2023-04-10T00:38:50.837433Z","shell.execute_reply.started":"2023-04-10T00:38:50.829572Z","shell.execute_reply":"2023-04-10T00:38:50.836233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_splitted['label'].value_counts()[:1].index.tolist()[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.839153Z","iopub.execute_input":"2023-04-10T00:38:50.839843Z","iopub.status.idle":"2023-04-10T00:38:50.852860Z","shell.execute_reply.started":"2023-04-10T00:38:50.839771Z","shell.execute_reply":"2023-04-10T00:38:50.851392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_splitted = test_data_splitted.rename(columns={'session_id': 'input'})","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.854971Z","iopub.execute_input":"2023-04-10T00:38:50.855364Z","iopub.status.idle":"2023-04-10T00:38:50.866633Z","shell.execute_reply.started":"2023-04-10T00:38:50.855328Z","shell.execute_reply":"2023-04-10T00:38:50.865175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_splitted","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.868851Z","iopub.execute_input":"2023-04-10T00:38:50.869299Z","iopub.status.idle":"2023-04-10T00:38:50.882303Z","shell.execute_reply.started":"2023-04-10T00:38:50.869263Z","shell.execute_reply":"2023-04-10T00:38:50.881323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### monitoring the memory usage\ngc.collect()\nsys_stats()","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:50.883874Z","iopub.execute_input":"2023-04-10T00:38:50.884525Z","iopub.status.idle":"2023-04-10T00:38:51.006374Z","shell.execute_reply.started":"2023-04-10T00:38:50.884488Z","shell.execute_reply":"2023-04-10T00:38:51.005092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# the zero_rule baseline model","metadata":{}},{"cell_type":"code","source":"def baseline_model_zero_rule(train_data_in, test_data_in):\n   zero_rule_highest_count = train_data_in['label'].value_counts()[:1].index.tolist()[0]\n   predicted = []\n   for index, test_row in test_data_in.iterrows():\n       predicted.append({ 'x': f\"{test_row['input']}\", 'y': zero_rule_highest_count })\n   return predicted","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:51.008286Z","iopub.execute_input":"2023-04-10T00:38:51.008634Z","iopub.status.idle":"2023-04-10T00:38:51.015838Z","shell.execute_reply.started":"2023-04-10T00:38:51.008599Z","shell.execute_reply":"2023-04-10T00:38:51.014729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# test the prediction on splitted dataset","metadata":{}},{"cell_type":"code","source":"%%time\ny = baseline_model_zero_rule(train_data_splitted, test_data_splitted)\ny_df = pd.DataFrame(y)\ny_df = y_df.rename(columns={'x': 'session_id', 'y': 'correct'})","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:51.017900Z","iopub.execute_input":"2023-04-10T00:38:51.018312Z","iopub.status.idle":"2023-04-10T00:38:55.497305Z","shell.execute_reply.started":"2023-04-10T00:38:51.018272Z","shell.execute_reply":"2023-04-10T00:38:55.496033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_df","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:55.499104Z","iopub.execute_input":"2023-04-10T00:38:55.499473Z","iopub.status.idle":"2023-04-10T00:38:55.513035Z","shell.execute_reply.started":"2023-04-10T00:38:55.499438Z","shell.execute_reply":"2023-04-10T00:38:55.511661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# f1_score","metadata":{}},{"cell_type":"code","source":"import sklearn\n#TODO sklearn.metrics.f1_score(true, pred, average='macro')\nf1_score = sklearn.metrics.f1_score(test_data_splitted['label'].values, y_df['correct'].values, average='macro')\n","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:55.514994Z","iopub.execute_input":"2023-04-10T00:38:55.515399Z","iopub.status.idle":"2023-04-10T00:38:55.549142Z","shell.execute_reply.started":"2023-04-10T00:38:55.515359Z","shell.execute_reply":"2023-04-10T00:38:55.547879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_score","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:38:55.550855Z","iopub.execute_input":"2023-04-10T00:38:55.551776Z","iopub.status.idle":"2023-04-10T00:38:55.558873Z","shell.execute_reply.started":"2023-04-10T00:38:55.551694Z","shell.execute_reply":"2023-04-10T00:38:55.557867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# prediction and submission","metadata":{}},{"cell_type":"code","source":"import jo_wilder\nif DEBUG:\n   jo_wilder.make_env.__called__ = False","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:39:32.842736Z","iopub.execute_input":"2023-04-10T00:39:32.843251Z","iopub.status.idle":"2023-04-10T00:39:32.849418Z","shell.execute_reply.started":"2023-04-10T00:39:32.843208Z","shell.execute_reply":"2023-04-10T00:39:32.847954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"env = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:39:34.262390Z","iopub.execute_input":"2023-04-10T00:39:34.263441Z","iopub.status.idle":"2023-04-10T00:39:34.269130Z","shell.execute_reply.started":"2023-04-10T00:39:34.263393Z","shell.execute_reply":"2023-04-10T00:39:34.268019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# note\n\n* this api will change from [for (sample_submission, test_data) in iter_test:]\n* to [for (test_data, sample_submission) in iter_test:]","metadata":{}},{"cell_type":"code","source":"counter = 0\n# The API will deliver two dataframes in this specific order,\n# for every session+level grouping (one group per session for each checkpoint)\nfor (sample_submission, test_data) in iter_test:\n    if counter == 0:\n        #log.info(f'sample_submission {sample_submission.head()}')\n        log.info(f'sample_submission shape {sample_submission.shape}')\n        #log.info(f'test.head {test_data.head()}')\n        log.info(f'test shape {test_data.shape}')\n        \n    test_data = test_data.rename(columns={'session_id': 'input'})\n    \n    ### prediction\n    y = baseline_model_zero_rule(train_data, test_data)\n    \n    ### build submission\n    y_df = pd.DataFrame(y)\n    y_df = y_df.rename(columns={'x': 'session_id', 'y': 'correct'})\n    sample_submission = y_df\n    log.info(f'sample_submission\\n {sample_submission}')\n    ## env.predict appends the session+level sample_submission to the overall\n    ## submission\n    env.predict(sample_submission)\n    counter += 1","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:39:36.364877Z","iopub.execute_input":"2023-04-10T00:39:36.365334Z","iopub.status.idle":"2023-04-10T00:39:36.438484Z","shell.execute_reply.started":"2023-04-10T00:39:36.365294Z","shell.execute_reply":"2023-04-10T00:39:36.437388Z"},"trusted":true},"execution_count":null,"outputs":[]}]}