{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import f1_score\nimport dask.dataframe as dd\nfrom sklearn.ensemble import RandomForestClassifier, VotingClassifier\nfrom xgboost import XGBClassifier\nimport gc\nimport numpy as np\nfrom dask.diagnostics import ProgressBar","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-05T21:36:00.123033Z","iopub.execute_input":"2023-03-05T21:36:00.123561Z","iopub.status.idle":"2023-03-05T21:36:00.128956Z","shell.execute_reply.started":"2023-03-05T21:36:00.123524Z","shell.execute_reply":"2023-03-05T21:36:00.128181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The competition has a computational constraint of 2 CPUs and 8GB of RAM, so we'll need to optimize your code for efficiency. Also, the competition encourages small and lightweight models, so we'll need to prioritize model efficiency over accuracy.","metadata":{}},{"cell_type":"code","source":"train_labels_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntrain_labels_df['session'] = train_labels_df.session_id.apply(lambda x: int(x.split('_')[0]) )\ntrain_labels_df['new_session_id_quantity'] = train_labels_df['session_id'].apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( train_labels_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-03-05T21:31:14.426454Z","iopub.execute_input":"2023-03-05T21:31:14.426846Z","iopub.status.idle":"2023-03-05T21:31:15.058069Z","shell.execute_reply.started":"2023-03-05T21:31:14.426813Z","shell.execute_reply":"2023-03-05T21:31:15.057201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print( train_labels_df.head)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\ntest_df =  pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv')\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-05T21:33:56.270137Z","iopub.execute_input":"2023-03-05T21:33:56.270516Z","iopub.status.idle":"2023-03-05T21:34:53.408190Z","shell.execute_reply.started":"2023-03-05T21:33:56.270481Z","shell.execute_reply":"2023-03-05T21:34:53.407099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def optimize_memory_usage(df):\n    \"\"\"\n    Optimize the memory usage of a Pandas dataframe by changing the data types of its columns.\n\n    Parameters:\n        df (pd.DataFrame): The dataframe to optimize.\n\n    Returns:\n        pd.DataFrame: The optimized dataframe.\n    \"\"\"\n    # Get the initial memory usage of the dataframe\n    initial_memory = df.memory_usage().sum() / 1024**2\n    \n    # Iterate through each column of the dataframe\n    for col in df.columns:\n        # Get the data type of the column\n        col_type = df[col].dtype\n        \n        # Optimize the data type based on the range of values in the column\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min >= 0:\n                    if c_max < 255:\n                        df[col] = df[col].astype('uint8')\n                    elif c_max < 65535:\n                        df[col] = df[col].astype('uint16')\n                    elif c_max < 4294967295:\n                        df[col] = df[col].astype('uint32')\n                    else:\n                        df[col] = df[col].astype('uint64')\n                else:\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype('int8')\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype('int16')\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype('int32')\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype('int64')\n            else:\n                if c_min >= np.finfo(np.float16).min and c_max <= np.finfo(np.float16).max:\n                    df[col] = df[col].astype('float16')\n                elif c_min >= np.finfo(np.float32).min and c_max <= np.finfo(np.float32).max:\n                    df[col] = df[col].astype('float32')\n                else:\n                    df[col] = df[col].astype('float64')\n    \n    # Get the final memory usage of the dataframe\n    final_memory = df.memory_usage().sum() / 1024**2\n    \n    print(f\"Memory usage optimized from {initial_memory:.2f} MB to {final_memory:.2f} MB\")\n    \n    return df\n","metadata":{"execution":{"iopub.status.busy":"2023-03-05T21:35:34.918057Z","iopub.execute_input":"2023-03-05T21:35:34.920070Z","iopub.status.idle":"2023-03-05T21:35:34.938295Z","shell.execute_reply.started":"2023-03-05T21:35:34.920003Z","shell.execute_reply":"2023-03-05T21:35:34.937204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = optimize_memory_usage(train_df)\ntest_df = optimize_memory_usage(test_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-05T21:36:06.601282Z","iopub.execute_input":"2023-03-05T21:36:06.602179Z","iopub.status.idle":"2023-03-05T21:36:10.874577Z","shell.execute_reply.started":"2023-03-05T21:36:06.602136Z","shell.execute_reply":"2023-03-05T21:36:10.873605Z"},"trusted":true},"execution_count":null,"outputs":[]}]}