{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Exporatory Data Analisys","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:03.634854Z","iopub.execute_input":"2023-04-03T18:19:03.635188Z","iopub.status.idle":"2023-04-03T18:19:03.687131Z","shell.execute_reply.started":"2023-04-03T18:19:03.635161Z","shell.execute_reply":"2023-04-03T18:19:03.685783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/predict-student-performance-from-game-play'","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:04.383238Z","iopub.execute_input":"2023-04-03T18:19:04.384002Z","iopub.status.idle":"2023-04-03T18:19:04.388899Z","shell.execute_reply.started":"2023-04-03T18:19:04.383965Z","shell.execute_reply":"2023-04-03T18:19:04.387380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:05.164227Z","iopub.execute_input":"2023-04-03T18:19:05.164975Z","iopub.status.idle":"2023-04-03T18:19:05.169496Z","shell.execute_reply.started":"2023-04-03T18:19:05.164936Z","shell.execute_reply":"2023-04-03T18:19:05.168174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Files Structure Overview","metadata":{}},{"cell_type":"code","source":"!ls -lah /kaggle/input/predict-student-performance-from-game-play","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:05.499917Z","iopub.execute_input":"2023-04-03T18:19:05.500611Z","iopub.status.idle":"2023-04-03T18:19:05.773980Z","shell.execute_reply.started":"2023-04-03T18:19:05.500576Z","shell.execute_reply":"2023-04-03T18:19:05.772398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### helper funcions","metadata":{}},{"cell_type":"code","source":"def file_info(file_name: str, base_path: str) -> None:\n    \"\"\"\n    Displays information about a CSV file located at the specified file path.\n\n    Parameters:\n    file_name (str): The name of the CSV file.\n    base_path (str): The base path of the CSV file.\n\n    Returns:\n    None\n    \"\"\"\n    # Construct full file path\n    file_path = f'{base_path}/{file_name}'\n\n    # Count number of rows in file\n    with open(file_path) as file:\n        num_rows = sum(1 for line in file)\n\n    # Display number of rows in file\n    print('Number of Rows:', num_rows)\n\n    # Display raw content of file\n    print('\\nRaw Content View:\\n')\n    with open(file_path, 'r') as file:\n        # Display first three lines\n        for i, line in enumerate(file):\n            print(line.strip())\n            if i == 2:\n                break\n\n        # Calculate file size and display last three lines\n        file.seek(0, 2)\n        file_size = file.tell()\n        file.seek(max(file_size-1024, 0))\n        for line in file.readlines()[-3:]:\n            print(line.strip())\n\n    # Display dataframe view of file\n    print('\\nDataframe View:\\n')\n    df = pd.read_csv(file_path, nrows=10)\n    display(df)\n\n    # Display sample view of file\n    print('\\nSample View:\\n')\n    sample_df = pd.concat([df.head(1), df.tail(1)], ignore_index=True)\n    display(sample_df.to_dict('records'))\n","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:05.797232Z","iopub.execute_input":"2023-04-03T18:19:05.797582Z","iopub.status.idle":"2023-04-03T18:19:05.807918Z","shell.execute_reply.started":"2023-04-03T18:19:05.797555Z","shell.execute_reply":"2023-04-03T18:19:05.806993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### train.csv","metadata":{}},{"cell_type":"code","source":"file_info('train.csv', BASE_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:06.104059Z","iopub.execute_input":"2023-04-03T18:19:06.104882Z","iopub.status.idle":"2023-04-03T18:19:58.621546Z","shell.execute_reply.started":"2023-04-03T18:19:06.104851Z","shell.execute_reply":"2023-04-03T18:19:58.620607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### test.csv","metadata":{}},{"cell_type":"code","source":"file_info('test.csv', BASE_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:58.623211Z","iopub.execute_input":"2023-04-03T18:19:58.623679Z","iopub.status.idle":"2023-04-03T18:19:58.673268Z","shell.execute_reply.started":"2023-04-03T18:19:58.623651Z","shell.execute_reply":"2023-04-03T18:19:58.672088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### train_labels","metadata":{}},{"cell_type":"code","source":"file_info('train_labels.csv', BASE_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:58.675928Z","iopub.execute_input":"2023-04-03T18:19:58.678076Z","iopub.status.idle":"2023-04-03T18:19:58.822490Z","shell.execute_reply.started":"2023-04-03T18:19:58.678044Z","shell.execute_reply":"2023-04-03T18:19:58.821661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### sample_submission","metadata":{}},{"cell_type":"code","source":"file_info('sample_submission.csv', BASE_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:58.824631Z","iopub.execute_input":"2023-04-03T18:19:58.825143Z","iopub.status.idle":"2023-04-03T18:19:58.849842Z","shell.execute_reply.started":"2023-04-03T18:19:58.825112Z","shell.execute_reply":"2023-04-03T18:19:58.849121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Room and screen position time serie","metadata":{}},{"cell_type":"markdown","source":"### helper functions","metadata":{}},{"cell_type":"code","source":"import json\nimport matplotlib.pyplot as plt\n\ndef plot_coordinates(json_data, show_labels=False, show_screen=False):\n    \"\"\"\n    Plots the coordinates for 'room_coor_x' and/or 'screen_coor_x' from the JSON data\n    with different colors for each set of coordinates. The points are sorted by index,\n    and the size of each point is proportional to the difference between the current\n    elapsed time and the last one. Only data points whose event_name ends with 'click'\n    are included in the plot.\n\n    Parameters:\n    json_data (list): A list of dictionaries containing the JSON data.\n    show_labels (bool): Whether to show labels for each point (default False).\n    show_screen (bool): Whether to show screen coordinates instead of room coordinates (default False).\n\n    Returns:\n    None\n    \"\"\"\n    # Initialize empty lists for x and y coordinates, labels, and volumes\n    x = []\n    y = []\n    labels = []\n    volumes = []\n\n    # Initialize the previous elapsed time value\n    prev_elapsed_time = 0\n\n    # Choose which coordinates to use based on show_screen parameter\n    if show_screen:\n        x_coord = 'screen_coor_x'\n        y_coord = 'screen_coor_y'\n    else:\n        x_coord = 'room_coor_x'\n        y_coord = 'room_coor_y'\n\n    # Iterate through JSON data to get relevant data points\n    for data in sorted(json_data, key=lambda x: x['index']):\n        if data['event_name'].endswith('click'):\n            x.append(data[x_coord])\n            y.append(data[y_coord])\n            if show_labels:\n                label = data['text'][:30].strip() if isinstance(data['text'], str) else '' # Trim label text to max 30 chars\n                labels.append(label)\n            current_elapsed_time = data['elapsed_time']\n            volumes.append(current_elapsed_time - prev_elapsed_time)\n            prev_elapsed_time = current_elapsed_time\n\n    # Calculate relative volumes and plot the coordinates with different colors and labels\n    max_volume = max(volumes)\n    volumes = [v / max_volume * 1000 for v in volumes]\n    plt.scatter(x, y, c='r' if not show_screen else 'b', label='Room' if not show_screen else 'Screen', s=volumes)\n    if show_labels:\n        for i, label in enumerate(labels):\n            plt.annotate(label, (x[i], y[i]), fontsize=8)\n\n    plt.legend()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:58.850922Z","iopub.execute_input":"2023-04-03T18:19:58.851331Z","iopub.status.idle":"2023-04-03T18:19:58.861330Z","shell.execute_reply.started":"2023-04-03T18:19:58.851290Z","shell.execute_reply":"2023-04-03T18:19:58.860361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(f'{BASE_PATH}/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:58.862582Z","iopub.execute_input":"2023-04-03T18:19:58.863093Z","iopub.status.idle":"2023-04-03T18:19:58.890823Z","shell.execute_reply.started":"2023-04-03T18:19:58.863059Z","shell.execute_reply":"2023-04-03T18:19:58.889161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session_ids = list(df_test.session_id.unique())","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:58.893268Z","iopub.execute_input":"2023-04-03T18:19:58.893919Z","iopub.status.idle":"2023-04-03T18:19:58.903878Z","shell.execute_reply.started":"2023-04-03T18:19:58.893888Z","shell.execute_reply":"2023-04-03T18:19:58.902639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[df_test['session_id']==session_ids[0]]","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:58.905023Z","iopub.execute_input":"2023-04-03T18:19:58.905507Z","iopub.status.idle":"2023-04-03T18:19:58.944638Z","shell.execute_reply.started":"2023-04-03T18:19:58.905476Z","shell.execute_reply":"2023-04-03T18:19:58.943151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_coordinates(df_test[df_test['session_id']==session_ids[0]].to_dict('records'))","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:58.946187Z","iopub.execute_input":"2023-04-03T18:19:58.946603Z","iopub.status.idle":"2023-04-03T18:19:59.176611Z","shell.execute_reply.started":"2023-04-03T18:19:58.946571Z","shell.execute_reply":"2023-04-03T18:19:59.175700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_coordinates(df_test[df_test['session_id']==session_ids[0]].to_dict('records'), show_labels=False, show_screen=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:59.178787Z","iopub.execute_input":"2023-04-03T18:19:59.179864Z","iopub.status.idle":"2023-04-03T18:19:59.384433Z","shell.execute_reply.started":"2023-04-03T18:19:59.179801Z","shell.execute_reply":"2023-04-03T18:19:59.383660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"import random\nimport pandas as pd\n\ndef randomize_correct(csv_path, new_csv_path):\n    \"\"\"\n    Reads a CSV file with the structure 'session_id,correct,session_level', and\n    randomly chooses 0 or 1 for the 'correct' field of each row. Saves the modified\n    DataFrame to a new CSV file with the specified path.\n\n    Parameters:\n    csv_path (str): The path to the original CSV file.\n    new_csv_path (str): The path to the new CSV file to be created.\n\n    Returns:\n    None\n    \"\"\"\n    # Read the CSV file into a DataFrame\n    df = pd.read_csv(csv_path)\n\n    # Randomly choose 0 or 1 for the correct field\n    df['correct'] = [random.choice([0, 1]) for _ in range(len(df))]\n\n    # Save the modified DataFrame to a new CSV file\n    df.to_csv(new_csv_path, index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:19:59.385876Z","iopub.execute_input":"2023-04-03T18:19:59.386419Z","iopub.status.idle":"2023-04-03T18:19:59.393521Z","shell.execute_reply.started":"2023-04-03T18:19:59.386383Z","shell.execute_reply":"2023-04-03T18:19:59.392079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"randomize_correct(f'{BASE_PATH}/sample_submission.csv', './my_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:21:54.873704Z","iopub.execute_input":"2023-04-03T18:21:54.874112Z","iopub.status.idle":"2023-04-03T18:21:54.885042Z","shell.execute_reply.started":"2023-04-03T18:21:54.874074Z","shell.execute_reply":"2023-04-03T18:21:54.883802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -lah","metadata":{"execution":{"iopub.status.busy":"2023-04-03T18:21:59.383056Z","iopub.execute_input":"2023-04-03T18:21:59.383460Z","iopub.status.idle":"2023-04-03T18:21:59.651863Z","shell.execute_reply.started":"2023-04-03T18:21:59.383425Z","shell.execute_reply":"2023-04-03T18:21:59.650511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}