{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ======================= \n# Import libraries \n# ======================= \nimport os\nimport pandas as pd\nimport numpy as np\npd.set_option('display.max_rows', 100)\npd.set_option('display.max_columns', 500)\npd.set_option('display.width', 1000)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom mpl_toolkits.mplot3d import Axes3D\nfrom plotly.offline import init_notebook_mode, iplot ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-07T14:45:59.360388Z","iopub.execute_input":"2023-02-07T14:45:59.360830Z","iopub.status.idle":"2023-02-07T14:45:59.367439Z","shell.execute_reply.started":"2023-02-07T14:45:59.360793Z","shell.execute_reply":"2023-02-07T14:45:59.366517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_df(df):\n    print(f\"shape of df : {df.shape}\")\n    display(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-02-07T14:38:32.857284Z","iopub.execute_input":"2023-02-07T14:38:32.857599Z","iopub.status.idle":"2023-02-07T14:38:32.862514Z","shell.execute_reply.started":"2023-02-07T14:38:32.857569Z","shell.execute_reply":"2023-02-07T14:38:32.861687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\nfrom pathlib import Path","metadata":{"execution":{"iopub.status.busy":"2023-02-07T14:38:32.863622Z","iopub.execute_input":"2023-02-07T14:38:32.864743Z","iopub.status.idle":"2023-02-07T14:38:32.889030Z","shell.execute_reply.started":"2023-02-07T14:38:32.864689Z","shell.execute_reply":"2023-02-07T14:38:32.887918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path  =  '../input/predict-student-performance-from-game-play/train.csv' \ntest_path  =  '../input/predict-student-performance-from-game-play/test.csv' \nlabels_path = '../input/predict-student-performance-from-game-play/train_labels.csv' \nsamplesubmission_path  = '../input/predict-student-performance-from-game-play/sample_submission.csv' ","metadata":{"execution":{"iopub.status.busy":"2023-02-07T14:53:34.956193Z","iopub.execute_input":"2023-02-07T14:53:34.956620Z","iopub.status.idle":"2023-02-07T14:53:34.961990Z","shell.execute_reply.started":"2023-02-07T14:53:34.956590Z","shell.execute_reply":"2023-02-07T14:53:34.960876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reduce Memory Usage\n# from https://www.kaggle.com/code/mohammad2012191/reduce-memory-usage-2gb-780mb\ndef reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-02-07T14:38:32.898479Z","iopub.execute_input":"2023-02-07T14:38:32.899388Z","iopub.status.idle":"2023-02-07T14:38:32.913915Z","shell.execute_reply.started":"2023-02-07T14:38:32.899351Z","shell.execute_reply":"2023-02-07T14:38:32.912693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train Dataset ","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(train_path) \ntrain = reduce_memory_usage(train)\nshow_df(train) \ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-07T14:46:06.804407Z","iopub.execute_input":"2023-02-07T14:46:06.804817Z","iopub.status.idle":"2023-02-07T14:47:00.536743Z","shell.execute_reply.started":"2023-02-07T14:46:06.804786Z","shell.execute_reply":"2023-02-07T14:47:00.535630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2023-02-07T14:39:57.832132Z","iopub.execute_input":"2023-02-07T14:39:57.832532Z","iopub.status.idle":"2023-02-07T14:40:12.855064Z","shell.execute_reply.started":"2023-02-07T14:39:57.832493Z","shell.execute_reply":"2023-02-07T14:40:12.854044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### train_labels","metadata":{}},{"cell_type":"code","source":"label_df = pd.read_csv(labels_path) \nlabel_df = reduce_memory_usage(label_df)\nshow_df(label_df) \nlabel_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-07T14:55:12.172735Z","iopub.execute_input":"2023-02-07T14:55:12.173177Z","iopub.status.idle":"2023-02-07T14:55:12.753249Z","shell.execute_reply.started":"2023-02-07T14:55:12.173136Z","shell.execute_reply":"2023-02-07T14:55:12.752125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hist of correct\nlabel_df['correct'].hist()","metadata":{"execution":{"iopub.status.busy":"2023-02-07T14:57:13.557448Z","iopub.execute_input":"2023-02-07T14:57:13.559317Z","iopub.status.idle":"2023-02-07T14:57:13.737618Z","shell.execute_reply.started":"2023-02-07T14:57:13.559247Z","shell.execute_reply":"2023-02-07T14:57:13.736639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test Dataset ","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(test_path) \ntest = reduce_memory_usage(test)\nshow_df(test) \ntest.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-07T14:47:00.538598Z","iopub.execute_input":"2023-02-07T14:47:00.539051Z","iopub.status.idle":"2023-02-07T14:47:00.630326Z","shell.execute_reply.started":"2023-02-07T14:47:00.539012Z","shell.execute_reply":"2023-02-07T14:47:00.629256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.describe()","metadata":{"execution":{"iopub.status.busy":"2023-02-07T14:40:12.960361Z","iopub.execute_input":"2023-02-07T14:40:12.960682Z","iopub.status.idle":"2023-02-07T14:40:13.019002Z","shell.execute_reply.started":"2023-02-07T14:40:12.960653Z","shell.execute_reply":"2023-02-07T14:40:13.018232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['label'] = 'train' \ntest['label']  = 'test' \n\ndf_all = pd.concat([train, test], axis=0)\ndf_all.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-02-07T14:48:25.722344Z","iopub.execute_input":"2023-02-07T14:48:25.723586Z","iopub.status.idle":"2023-02-07T14:48:27.421296Z","shell.execute_reply.started":"2023-02-07T14:48:25.723534Z","shell.execute_reply":"2023-02-07T14:48:27.420146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission CSV","metadata":{}},{"cell_type":"code","source":"sub_df = pd.read_csv(samplesubmission_path) \nshow_df(sub_df) \nsub_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-07T14:52:28.958083Z","iopub.execute_input":"2023-02-07T14:52:28.958480Z","iopub.status.idle":"2023-02-07T14:52:28.984065Z","shell.execute_reply.started":"2023-02-07T14:52:28.958448Z","shell.execute_reply":"2023-02-07T14:52:28.983015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 列の型を取得するための変数\ncol_types = df_all.dtypes\n\n# 空のリストを作成する\nnumeric_cols = []\nobject_cols = []\n\n# カラム毎に中身を軽く分析 \nfor col, col_type in col_types.items():\n    if col_type == 'object':\n        object_cols.append(col)\n        ratio = df_all[col].nunique() / len(df_all)\n        ave_len = df_all[col].str.len().mean()\n        print(f'{col}: object (unique ratio {ratio:.3f}, average length {ave_len:.0f})')\n    else:\n        numeric_cols.append(col)\n        print('%s: numeric' % col)","metadata":{"execution":{"iopub.status.busy":"2023-02-07T14:49:05.557932Z","iopub.execute_input":"2023-02-07T14:49:05.558389Z","iopub.status.idle":"2023-02-07T14:49:25.132179Z","shell.execute_reply.started":"2023-02-07T14:49:05.558349Z","shell.execute_reply":"2023-02-07T14:49:25.130863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}