{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd \n ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-12T09:31:49.179298Z","iopub.execute_input":"2023-05-12T09:31:49.179652Z","iopub.status.idle":"2023-05-12T09:31:49.185779Z","shell.execute_reply.started":"2023-05-12T09:31:49.179620Z","shell.execute_reply":"2023-05-12T09:31:49.184798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndataframe = pd.read_csv('../input/predict-student-performance-from-game-play/train.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-05-12T09:31:52.775877Z","iopub.execute_input":"2023-05-12T09:31:52.776313Z","iopub.status.idle":"2023-05-12T09:33:48.880513Z","shell.execute_reply.started":"2023-05-12T09:31:52.776277Z","shell.execute_reply":"2023-05-12T09:33:48.879465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T10:07:41.390924Z","iopub.execute_input":"2023-05-12T10:07:41.391281Z","iopub.status.idle":"2023-05-12T10:07:41.508325Z","shell.execute_reply.started":"2023-05-12T10:07:41.391250Z","shell.execute_reply":"2023-05-12T10:07:41.507364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def delete_columns_and_save(dataframe, column_names,column_name, save_path):\n    # Check if the columns exist in the dataframe\n    columns_not_found = [col for col in column_names if col not in dataframe.columns]\n    if columns_not_found:\n        print(f\"Columns {columns_not_found} do not exist in the dataframe.\")\n        return\n\n    # Delete the specified columns\n    dataframe.drop(column_names, axis=1, inplace=True)\n    print(f\"Columns {column_names} have been deleted from the dataframe.\")\n    # Convert the values from milliseconds to seconds\n    dataframe[column_name] = dataframe[column_name] / 1000.0\n    print(f\"Column '{column_name}' has been modified from milliseconds to seconds.\")\n\n\n    # Save the updated dataframe to a new file\n    dataframe.to_pickle(save_path)\n    print(f\"The updated dataframe has been saved to '{save_path}'.\")\ndelete_columns_and_save(dataframe, ['hover_duration', 'fullscreen', 'hq','music'],'elapsed_time', 'updated_data.pickle')\n","metadata":{"execution":{"iopub.status.busy":"2023-05-12T10:07:44.698366Z","iopub.execute_input":"2023-05-12T10:07:44.698708Z","iopub.status.idle":"2023-05-12T10:08:02.807467Z","shell.execute_reply.started":"2023-05-12T10:07:44.698680Z","shell.execute_reply":"2023-05-12T10:08:02.806381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ngc.collect()\ndel dataframe","metadata":{"execution":{"iopub.status.busy":"2023-05-12T10:09:44.649589Z","iopub.execute_input":"2023-05-12T10:09:44.650582Z","iopub.status.idle":"2023-05-12T10:09:44.784936Z","shell.execute_reply.started":"2023-05-12T10:09:44.650539Z","shell.execute_reply":"2023-05-12T10:09:44.783712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data= pd.read_pickle('../working/updated_data.pickle')","metadata":{"execution":{"iopub.status.busy":"2023-05-12T10:10:37.380584Z","iopub.execute_input":"2023-05-12T10:10:37.381292Z","iopub.status.idle":"2023-05-12T10:10:49.465019Z","shell.execute_reply.started":"2023-05-12T10:10:37.381256Z","shell.execute_reply":"2023-05-12T10:10:49.464006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info(memory_usage=\"deep\")","metadata":{"execution":{"iopub.status.busy":"2023-05-12T10:11:13.627814Z","iopub.execute_input":"2023-05-12T10:11:13.628796Z","iopub.status.idle":"2023-05-12T10:11:51.475771Z","shell.execute_reply.started":"2023-05-12T10:11:13.628741Z","shell.execute_reply":"2023-05-12T10:11:51.474750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T10:12:00.649331Z","iopub.execute_input":"2023-05-12T10:12:00.649686Z","iopub.status.idle":"2023-05-12T10:12:00.772766Z","shell.execute_reply.started":"2023-05-12T10:12:00.649656Z","shell.execute_reply":"2023-05-12T10:12:00.771716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['session_id'] = data['session_id'].astype('int32')\ndata['index'] = data['index'].astype('int32')\ndata['elapsed_time']= data['elapsed_time'].astype('float32')\ndata['event_name']= data['event_name'].astype('category')\ndata['name']= data['name'].astype('category')\ndata['level']=data['level'].astype('int32')\ndata['page']=data['page'].astype('float32')\ndata['room_coor_x']= data['room_coor_x'].astype('float32')\ndata['room_coor_y']= data['room_coor_y'].astype('float32')\ndata['screen_coor_x'] = data['screen_coor_x'].astype('float32')\ndata['screen_coor_y'] = data['screen_coor_y'].astype('float32')\ndata['text'] = data['text'].astype('category')   \ndata['fqid'] = data['fqid'].astype('category')\ndata['room_fqid'] = data['room_fqid'].astype('category')\ndata['text_fqid'] = data['text_fqid'].astype('category')\ndata['level_group'] = data['level_group'].astype('category') ","metadata":{"execution":{"iopub.status.busy":"2023-05-12T10:12:04.024637Z","iopub.execute_input":"2023-05-12T10:12:04.025238Z","iopub.status.idle":"2023-05-12T10:12:34.897999Z","shell.execute_reply.started":"2023-05-12T10:12:04.025204Z","shell.execute_reply":"2023-05-12T10:12:34.897026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.to_pickle('output_pickle_file.pickle')","metadata":{"execution":{"iopub.status.busy":"2023-05-12T10:12:41.121668Z","iopub.execute_input":"2023-05-12T10:12:41.122039Z","iopub.status.idle":"2023-05-12T10:12:42.112855Z","shell.execute_reply.started":"2023-05-12T10:12:41.122009Z","shell.execute_reply":"2023-05-12T10:12:42.111805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info(memory_usage=\"deep\")","metadata":{"execution":{"iopub.status.busy":"2023-05-12T10:12:46.523274Z","iopub.execute_input":"2023-05-12T10:12:46.523622Z","iopub.status.idle":"2023-05-12T10:12:46.541380Z","shell.execute_reply.started":"2023-05-12T10:12:46.523592Z","shell.execute_reply":"2023-05-12T10:12:46.540380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data2 = pd.read_pickle('/kaggle/working/output_pickle_file.pickle')","metadata":{"execution":{"iopub.status.busy":"2023-05-12T10:12:50.199472Z","iopub.execute_input":"2023-05-12T10:12:50.199849Z","iopub.status.idle":"2023-05-12T10:12:50.859456Z","shell.execute_reply.started":"2023-05-12T10:12:50.199818Z","shell.execute_reply":"2023-05-12T10:12:50.858472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data","metadata":{"execution":{"iopub.status.busy":"2023-05-12T10:12:54.844545Z","iopub.execute_input":"2023-05-12T10:12:54.845420Z","iopub.status.idle":"2023-05-12T10:12:54.856650Z","shell.execute_reply.started":"2023-05-12T10:12:54.845374Z","shell.execute_reply":"2023-05-12T10:12:54.855559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data2.info(memory_usage=\"deep\")","metadata":{"execution":{"iopub.status.busy":"2023-05-12T10:13:07.037167Z","iopub.execute_input":"2023-05-12T10:13:07.037551Z","iopub.status.idle":"2023-05-12T10:13:07.053476Z","shell.execute_reply.started":"2023-05-12T10:13:07.037517Z","shell.execute_reply":"2023-05-12T10:13:07.052384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"column = 'level_group'\nsplit_value = data2[column].unique()\nfor value in split_value :\n    data_splited= data2[data2[column]==value]\n    data_splited.to_csv(f'{value}.csv', index=False)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-12T10:25:37.059910Z","iopub.execute_input":"2023-05-12T10:25:37.060880Z","iopub.status.idle":"2023-05-12T10:30:40.571313Z","shell.execute_reply.started":"2023-05-12T10:25:37.060828Z","shell.execute_reply":"2023-05-12T10:30:40.570250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas_profiling as pp","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pp.ProfileReport(data2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def verify(dataframe, column_name, value):\n    if column_name not in dataframe.columns:\n        print(f\"Column '{column_name}' does not exist in the dataframe.\")\n        return\n    if value in dataframe[column_name].values:\n        print(f\"Column '{column_name}' contains the value '{value}'.\")\n    else:\n        print(f\"Column '{column_name}' does not contain the value '{value}'.\")\nverify(dataframe, 'index', 0)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def verify_column_empty_values(dataframe, column_name):\n    # Check if the column exists in the dataframe\n    if column_name not in dataframe.columns:\n        print(f\"Column '{column_name}' does not exist in the dataframe.\")\n        return\n\n    # Check if the column contains any empty values\n    if dataframe[column_name].isnull().any():\n        print(f\"Column '{column_name}' contains empty values.\")\n    else:\n        print(f\"Column '{column_name}' does not contain empty values.\")\nverify_column_empty_values(dataframe, 'text_fqid')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def verify_column_numeric_value_repeated(dataframe, column_namee, value):\n    # Check if the column exists in the dataframe\n    if column_namee not in dataframe.columns:\n        print(f\"Column '{column_namee}' does not exist in the dataframe.\")\n        return\n\n    # Count the occurrences of the specified value in the column\n    value_count = dataframe[column_namee].value_counts().get(value, 0)\n\n    # Check if the value appears more than once\n    if value_count > 1:\n        print(f\"Column '{column_namee}' contains the numeric value '{value}' more than once.\")\n    else:\n        print(f\"Column '{column_namee}' does not contain the numeric value '{value}' more than once.\")\nverify_column_numeric_value_repeated(dataframe, 'index', 0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_column_values(dataframe, columnn_name):\n    # Check if the column exists in the dataframe\n    if columnn_name not in dataframe.columns:\n        print(f\"Column '{columnn_name}' does not exist in the dataframe.\")\n        return None\n\n    # Retrieve the values in the specified column\n    column_values = dataframe[columnn_name].values\n\n    return column_values\nname_column_values = get_column_values(dataframe, 'fqid')\nprint(name_column_values)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}