{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-19T13:31:22.733779Z","iopub.execute_input":"2023-04-19T13:31:22.734245Z","iopub.status.idle":"2023-04-19T13:31:22.74596Z","shell.execute_reply.started":"2023-04-19T13:31:22.734208Z","shell.execute_reply":"2023-04-19T13:31:22.744597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nplt.style.use('fivethirtyeight')\nplt.rcParams['figure.figsize'] = (30, 20)\nsns.set_style('darkgrid')","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:31:26.639388Z","iopub.execute_input":"2023-04-19T13:31:26.639898Z","iopub.status.idle":"2023-04-19T13:31:27.685418Z","shell.execute_reply.started":"2023-04-19T13:31:26.639855Z","shell.execute_reply":"2023-04-19T13:31:27.684161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\", usecols=[0])\ndisplay(tmp)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T06:55:15.726134Z","iopub.execute_input":"2023-04-17T06:55:15.727344Z","iopub.status.idle":"2023-04-17T06:55:46.447889Z","shell.execute_reply.started":"2023-04-17T06:55:15.727285Z","shell.execute_reply":"2023-04-17T06:55:46.446807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = tmp.groupby('session_id').session_id.agg('count')\nprint(tmp)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T06:56:09.565293Z","iopub.execute_input":"2023-04-17T06:56:09.565735Z","iopub.status.idle":"2023-04-17T06:56:10.147687Z","shell.execute_reply.started":"2023-04-17T06:56:09.565698Z","shell.execute_reply":"2023-04-17T06:56:10.146756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp.iloc[0:5]","metadata":{"execution":{"iopub.status.busy":"2023-04-17T06:56:13.814975Z","iopub.execute_input":"2023-04-17T06:56:13.815379Z","iopub.status.idle":"2023-04-17T06:56:13.823736Z","shell.execute_reply.started":"2023-04-17T06:56:13.81534Z","shell.execute_reply":"2023-04-17T06:56:13.822732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.enable()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-17T06:56:17.908528Z","iopub.execute_input":"2023-04-17T06:56:17.908955Z","iopub.status.idle":"2023-04-17T06:56:18.521755Z","shell.execute_reply.started":"2023-04-17T06:56:17.90892Z","shell.execute_reply":"2023-04-17T06:56:18.520844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ITER = 20\nPIECES = int(np.ceil(len(tmp) / ITER))\n\nreads = []\nskips = [0]\nfor k in range(ITER) :\n    a = k*PIECES\n    b = (k+1)*PIECES\n    if b>len(tmp) : b=len(tmp)\n    r = tmp.iloc[a:b].sum()\n    reads.append(r)\n    skips.append(skips[-1] + r)\nprint(f'untuk menghindari kesalahan memori, kita akan membaca kereta dalam {PIECES} ukuran potongan:')\nprint(reads)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T06:56:22.972112Z","iopub.execute_input":"2023-04-17T06:56:22.97251Z","iopub.status.idle":"2023-04-17T06:56:22.983875Z","shell.execute_reply.started":"2023-04-17T06:56:22.972474Z","shell.execute_reply":"2023-04-17T06:56:22.982672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_id = pl.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', columns = ['index'])\ndisplay(train_id)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T07:20:43.893382Z","iopub.execute_input":"2023-04-17T07:20:43.893786Z","iopub.status.idle":"2023-04-17T07:21:01.938149Z","shell.execute_reply.started":"2023-04-17T07:20:43.893753Z","shell.execute_reply":"2023-04-17T07:21:01.937184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_train = len(train_id)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T07:21:24.564295Z","iopub.execute_input":"2023-04-17T07:21:24.564815Z","iopub.status.idle":"2023-04-17T07:21:24.569883Z","shell.execute_reply.started":"2023-04-17T07:21:24.564779Z","shell.execute_reply":"2023-04-17T07:21:24.568714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.enable()\ndel train_id\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-17T07:21:27.198504Z","iopub.execute_input":"2023-04-17T07:21:27.199595Z","iopub.status.idle":"2023-04-17T07:21:27.64964Z","shell.execute_reply.started":"2023-04-17T07:21:27.199551Z","shell.execute_reply":"2023-04-17T07:21:27.648565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtyping = {\n    'session_id' : np.uint64,\n    'index' : np.uint16,\n    'elapsed_time' : np.uint16,\n    'event_name' : 'category',\n    'name' : 'category',\n    'level' : np.uint8,\n    'page' : np.uint16,\n    'room_coor_x' : np.float16,\n    'room_coor_y' : np.float16,\n    'screen_coor_x' : np.float16,\n    'screen_coor_y' : np.float16,\n    'hover_duration' : np.float16,\n    'text' : 'category',\n    'fqid' : 'category',\n    'room_fqid' : 'category',\n    'text_fqid' : 'category',\n    'fullscreen' : np.bool8,\n    'hq' : np.bool8,\n    'music' : np.bool8,\n    'level_group' : 'category'\n}","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:34:43.926079Z","iopub.execute_input":"2023-04-19T13:34:43.926541Z","iopub.status.idle":"2023-04-19T13:34:43.934577Z","shell.execute_reply.started":"2023-04-19T13:34:43.926505Z","shell.execute_reply":"2023-04-19T13:34:43.933427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf0 = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', nrows = reads[0], dtype = dtyping)\nmem_usage = df0.memory_usage().sum() / 1024 ** 2\nprint(f'Usage Memory : {mem_usage} MB')","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:59:58.112395Z","iopub.execute_input":"2023-04-19T13:59:58.112792Z","iopub.status.idle":"2023-04-19T13:59:58.13483Z","shell.execute_reply.started":"2023-04-19T13:59:58.112759Z","shell.execute_reply":"2023-04-19T13:59:58.133583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df0","metadata":{"execution":{"iopub.status.busy":"2023-04-19T14:00:09.358774Z","iopub.execute_input":"2023-04-19T14:00:09.359277Z","iopub.status.idle":"2023-04-19T14:00:09.383143Z","shell.execute_reply.started":"2023-04-19T14:00:09.359234Z","shell.execute_reply":"2023-04-19T14:00:09.381451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_pieces = []\nfor k in range(ITER) :\n    print(k, ',', end = ' ')\n    train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', nrows = reads[k], skiprows = skips[k]) \n    all_pieces.append(train)\n    \nprint('\\n')\ndel train; gc.collect()\ntrain_df = pd.concat(all_pieces, axis = 0)\nprint(f'Shape of train DF : {train_df.shape}')\ndisplay(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:57:49.801159Z","iopub.execute_input":"2023-04-19T13:57:49.801572Z","iopub.status.idle":"2023-04-19T13:57:49.819372Z","shell.execute_reply.started":"2023-04-19T13:57:49.801538Z","shell.execute_reply":"2023-04-19T13:57:49.818338Z"},"trusted":true},"execution_count":null,"outputs":[]}]}