{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-18T12:30:38.106680Z","iopub.execute_input":"2023-03-18T12:30:38.107088Z","iopub.status.idle":"2023-03-18T12:30:38.118141Z","shell.execute_reply.started":"2023-03-18T12:30:38.107052Z","shell.execute_reply":"2023-03-18T12:30:38.116950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport seaborn as sns\nimport warnings\nimport gc\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-18T12:58:04.520314Z","iopub.execute_input":"2023-03-18T12:58:04.520736Z","iopub.status.idle":"2023-03-18T12:58:04.527517Z","shell.execute_reply.started":"2023-03-18T12:58:04.520699Z","shell.execute_reply":"2023-03-18T12:58:04.526197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem(df):\n    start_mem = df.memory_usage().sum() / 1024 ** 2\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n    end_mem = df.memory_usage().sum() / 1024 ** 2\n    print('{:.2f} Mb, {:.2f} Mb ({:.2f} %)'.format(start_mem, end_mem, 100 * (start_mem - end_mem) / start_mem))\n    gc.collect()\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-03-18T12:44:29.697336Z","iopub.execute_input":"2023-03-18T12:44:29.698173Z","iopub.status.idle":"2023-03-18T12:44:29.713333Z","shell.execute_reply.started":"2023-03-18T12:44:29.698118Z","shell.execute_reply":"2023-03-18T12:44:29.711730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-18T12:44:46.824040Z","iopub.execute_input":"2023-03-18T12:44:46.824483Z","iopub.status.idle":"2023-03-18T12:45:53.501995Z","shell.execute_reply.started":"2023-03-18T12:44:46.824430Z","shell.execute_reply":"2023-03-18T12:45:53.501007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = reduce_mem(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-18T12:49:56.837772Z","iopub.execute_input":"2023-03-18T12:49:56.838201Z","iopub.status.idle":"2023-03-18T12:50:00.315970Z","shell.execute_reply.started":"2023-03-18T12:49:56.838165Z","shell.execute_reply":"2023-03-18T12:50:00.314613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/test.csv\")\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-18T12:51:20.884134Z","iopub.execute_input":"2023-03-18T12:51:20.884570Z","iopub.status.idle":"2023-03-18T12:51:20.927049Z","shell.execute_reply.started":"2023-03-18T12:51:20.884533Z","shell.execute_reply":"2023-03-18T12:51:20.926065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\ntargets.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-18T12:51:53.252250Z","iopub.execute_input":"2023-03-18T12:51:53.253173Z","iopub.status.idle":"2023-03-18T12:51:53.492985Z","shell.execute_reply.started":"2023-03-18T12:51:53.253122Z","shell.execute_reply":"2023-03-18T12:51:53.491934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets.describe().T","metadata":{"execution":{"iopub.status.busy":"2023-03-18T12:52:41.459036Z","iopub.execute_input":"2023-03-18T12:52:41.459421Z","iopub.status.idle":"2023-03-18T12:52:41.499783Z","shell.execute_reply.started":"2023-03-18T12:52:41.459388Z","shell.execute_reply":"2023-03-18T12:52:41.498821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets[\"session\"] = targets[\"session_id\"].apply(lambda x:int(x.split('_')[0]))\ntargets[\"q\"] = targets[\"session_id\"].apply(lambda x:int(x.split('_')[1][1:]))\ntargets.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-18T12:53:16.809895Z","iopub.execute_input":"2023-03-18T12:53:16.810298Z","iopub.status.idle":"2023-03-18T12:53:17.206004Z","shell.execute_reply.started":"2023-03-18T12:53:16.810263Z","shell.execute_reply":"2023-03-18T12:53:17.204569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/sample_submission.csv\")\nsample","metadata":{"execution":{"iopub.status.busy":"2023-03-18T12:53:51.570914Z","iopub.execute_input":"2023-03-18T12:53:51.571332Z","iopub.status.idle":"2023-03-18T12:53:51.593425Z","shell.execute_reply.started":"2023-03-18T12:53:51.571295Z","shell.execute_reply":"2023-03-18T12:53:51.592221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info(verbose=True,null_counts=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-18T12:58:57.191920Z","iopub.execute_input":"2023-03-18T12:58:57.194081Z","iopub.status.idle":"2023-03-18T12:59:02.688339Z","shell.execute_reply.started":"2023-03-18T12:58:57.194028Z","shell.execute_reply":"2023-03-18T12:59:02.686891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print()\nprint('The Most Important Event Names in the data are:')\nprint()\nfor i,j in enumerate(['navigate_click','observation_click','notification_click',\n 'object_click','object_hover','map_hover','map_click','notebook_click','checkpoint']):\n    print(f'{i+1}.', j)","metadata":{"execution":{"iopub.status.busy":"2023-03-18T13:52:30.268543Z","iopub.execute_input":"2023-03-18T13:52:30.269036Z","iopub.status.idle":"2023-03-18T13:52:30.276771Z","shell.execute_reply.started":"2023-03-18T13:52:30.268996Z","shell.execute_reply":"2023-03-18T13:52:30.275737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}