{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":9925531,"sourceType":"datasetVersion","datasetId":6047696}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"import gc\nimport pandas as pd\nimport numpy as np\nimport datetime as dt\n\nimport matplotlib.pyplot as plt\nimport matplotlib.cm as cm\nimport seaborn as sns\n\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nimport plotly.express as px\nimport plotly.offline\n\nfrom colorama import Fore, Style, init\nfrom pprint import pprint\n\n# 🚫 Suppressing warnings 🚫\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-23T17:56:47.691595Z","iopub.execute_input":"2024-11-23T17:56:47.691999Z","iopub.status.idle":"2024-11-23T17:56:49.587781Z","shell.execute_reply.started":"2024-11-23T17:56:47.691964Z","shell.execute_reply":"2024-11-23T17:56:49.586375Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nimport optuna\nfrom sklearn.metrics import cohen_kappa_score, make_scorer, confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom imblearn.over_sampling import SMOTE\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom scipy.optimize import minimize\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-11-23T17:56:49.589853Z","iopub.execute_input":"2024-11-23T17:56:49.590839Z","iopub.status.idle":"2024-11-23T17:56:51.672864Z","shell.execute_reply.started":"2024-11-23T17:56:49.590789Z","shell.execute_reply":"2024-11-23T17:56:51.671631Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom tqdm.auto import tqdm \nfrom concurrent.futures import ThreadPoolExecutor\nfrom joblib import Parallel, delayed\nfrom time import sleep, time\nfrom multiprocessing import cpu_count\nimport polars as pl\nfrom sklearn.preprocessing import MinMaxScaler\nimport concurrent.futures\n\nfrom datetime import datetime, timezone, timedelta","metadata":{"execution":{"iopub.status.busy":"2024-11-23T17:56:51.674141Z","iopub.execute_input":"2024-11-23T17:56:51.674809Z","iopub.status.idle":"2024-11-23T17:56:51.980271Z","shell.execute_reply.started":"2024-11-23T17:56:51.674772Z","shell.execute_reply":"2024-11-23T17:56:51.979095Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load the data","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nlen(df_test)","metadata":{"execution":{"iopub.status.busy":"2024-11-23T17:56:51.983099Z","iopub.execute_input":"2024-11-23T17:56:51.983568Z","iopub.status.idle":"2024-11-23T17:56:52.005452Z","shell.execute_reply.started":"2024-11-23T17:56:51.983519Z","shell.execute_reply":"2024-11-23T17:56:52.004064Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-11-23T17:56:52.007490Z","iopub.execute_input":"2024-11-23T17:56:52.007875Z","iopub.status.idle":"2024-11-23T17:56:52.015796Z","shell.execute_reply.started":"2024-11-23T17:56:52.007838Z","shell.execute_reply":"2024-11-23T17:56:52.014598Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\nlen(test_ts)","metadata":{"execution":{"iopub.status.busy":"2024-11-23T17:56:52.017216Z","iopub.execute_input":"2024-11-23T17:56:52.017639Z","iopub.status.idle":"2024-11-23T17:56:52.441873Z","shell.execute_reply.started":"2024-11-23T17:56:52.017593Z","shell.execute_reply":"2024-11-23T17:56:52.440745Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ts['id'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T17:56:52.443428Z","iopub.execute_input":"2024-11-23T17:56:52.443896Z","iopub.status.idle":"2024-11-23T17:56:52.457224Z","shell.execute_reply.started":"2024-11-23T17:56:52.443848Z","shell.execute_reply":"2024-11-23T17:56:52.455870Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/colombian-frenchteam-problematicinternetusage/Dataset_problematic_internet_usage.csv')\nlen(df)","metadata":{"execution":{"iopub.status.busy":"2024-11-23T18:58:39.732991Z","iopub.execute_input":"2024-11-23T18:58:39.733383Z","iopub.status.idle":"2024-11-23T18:58:39.809342Z","shell.execute_reply.started":"2024-11-23T18:58:39.733352Z","shell.execute_reply":"2024-11-23T18:58:39.807968Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.groupby('Train_Test_Label').size()","metadata":{"execution":{"iopub.status.busy":"2024-11-23T18:58:42.137028Z","iopub.execute_input":"2024-11-23T18:58:42.137514Z","iopub.status.idle":"2024-11-23T18:58:42.147964Z","shell.execute_reply.started":"2024-11-23T18:58:42.137476Z","shell.execute_reply":"2024-11-23T18:58:42.146633Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df[(df['Train_Test_Label']=='test') | (df['Train_Test_Label']=='train') ]\nlen(df)","metadata":{"execution":{"iopub.status.busy":"2024-11-23T18:58:43.467872Z","iopub.execute_input":"2024-11-23T18:58:43.468250Z","iopub.status.idle":"2024-11-23T18:58:43.477989Z","shell.execute_reply.started":"2024-11-23T18:58:43.468220Z","shell.execute_reply":"2024-11-23T18:58:43.476932Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[df['sii']==3].head(3)","metadata":{"execution":{"iopub.status.busy":"2024-11-23T18:58:45.401795Z","iopub.execute_input":"2024-11-23T18:58:45.402161Z","iopub.status.idle":"2024-11-23T18:58:45.425982Z","shell.execute_reply.started":"2024-11-23T18:58:45.402131Z","shell.execute_reply":"2024-11-23T18:58:45.424926Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[df['sii']==0].head(3)","metadata":{"execution":{"iopub.status.busy":"2024-11-23T18:58:46.773103Z","iopub.execute_input":"2024-11-23T18:58:46.773588Z","iopub.status.idle":"2024-11-23T18:58:46.797985Z","shell.execute_reply.started":"2024-11-23T18:58:46.773553Z","shell.execute_reply":"2024-11-23T18:58:46.796534Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Plot functions 2","metadata":{}},{"cell_type":"code","source":"kid_id = '71ee31f8'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T17:59:13.349979Z","iopub.execute_input":"2024-11-23T17:59:13.350403Z","iopub.status.idle":"2024-11-23T17:59:13.355608Z","shell.execute_reply.started":"2024-11-23T17:59:13.350366Z","shell.execute_reply":"2024-11-23T17:59:13.354170Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feat_eng(df):\n\n    df['time_of_day_hours'] = (df['time_of_day'] / 1e9 / 3600)  # nanoseconds to hours\n    df['time_of_day_half_hours'] = (df['time_of_day'] / 1e9 / 1800)  # nanoseconds to half-hours\n    df['time_of_day_half_half_hours'] = (df['time_of_day'] / 1e9 / 900)  # nanoseconds to 15 minutes interval\n    df['time_of_day_fivemin_hours'] = (df['time_of_day'] / 1e9 / 300)  # nanoseconds to 5 minutes interval\n    df['day_time'] = df['relative_date_PCIAT'] + (df['time_of_day_hours'] / 24)\n    \n    # Day period assignment\n    day_start_hour = 8\n    day_end_hour = 21\n    df['day_period'] = np.where(\n            (df['time_of_day_hours'] >= day_start_hour) &\n            (df['time_of_day_hours'] < day_end_hour),\n            'day', 'night'\n        )\n    \n    # Initialize the 'which_day' column and day change detection\n    df['which_day'] = 0\n    day_change = (\n            (df['weekday'] != df['weekday'].shift(1)) |\n            (df['hour'] < df['hour'].shift(1)) |\n            ((df['hour'] == df['hour'].shift(1)) & (df['minute'] < df['minute'].shift(1))) |\n            ((df['hour'] == df['hour'].shift(1)) & (df['minute'] == df['minute'].shift(1)) & (df['second'] < df['second'].shift(1)))\n        )\n    df['which_day'] = day_change.cumsum() + 1\n    df['day_period_b'] = np.where(df['day_period'] == 'day', 1, 0)\n    df['time_of_day'] = pd.to_timedelta(df['time_of_day'], unit='ns')\n    base_date = pd.to_datetime('2024-01-01')\n    df['date'] = base_date + pd.to_timedelta(df['which_day'] - 1, unit='D')\n    df['timestamp'] = df['date'] + df['time_of_day']\n    df['timestamp'] = df['timestamp'].apply(lambda t: t.tz_localize(None))\n    df['timestamp_2'] = pd.to_datetime(df['timestamp']).apply(lambda t: t.tz_localize(None))\n    df.sort_values(['timestamp_2'], inplace=True)\n    df.set_index('timestamp_2', inplace=True)\n    \n    df[\"anglez\"] = df[\"anglez\"].astype(np.float32)\n    df[\"anglezdiffabs\"] = df[\"anglez\"].diff().abs().astype(np.float32)\n        \n    for col in ['anglezdiffabs']:\n            \n        # periods in seconds        \n        periods = [60] \n            \n        for n in periods:\n                \n            rol_args = {'window':f'{n+5}s', 'min_periods':10, 'center':True}\n                \n            for agg in ['median']:\n                df[f'{col}_{agg}_{n}'] = df[col].rolling(**rol_args).agg(agg).astype(np.float32).values\n                gc.collect()\n                \n            gc.collect()\n    \n    df.reset_index(inplace=True)\n    df.dropna(inplace=True)\n    df['large_enmo'] = df['enmo'] > 0.1509000062942505\n    df['anglezdiffabs_median_60_norm'] = (df['anglezdiffabs_median_60'] - np.min(df['anglezdiffabs_median_60'])) / (max(df['anglezdiffabs_median_60']) - min(df['anglezdiffabs_median_60']))\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:06:11.798561Z","iopub.execute_input":"2024-11-23T19:06:11.799339Z","iopub.status.idle":"2024-11-23T19:06:11.814416Z","shell.execute_reply.started":"2024-11-23T19:06:11.799301Z","shell.execute_reply":"2024-11-23T19:06:11.813264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feat_eng_by_id(idx):\n    \n    from warnings import simplefilter \n    simplefilter(action=\"ignore\", category=pd.errors.PerformanceWarning)\n    \n    df = (\n        pl.scan_parquet(f'/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id={idx}/part-0.parquet')\n        .with_columns(\n            (pl.col(\"time_of_day\").cast(pl.Int64) / 1_000_000_000).alias(\"total_seconds\")\n        )\n        .with_columns(\n            [\n                (pl.col(\"total_seconds\") // 3600).alias(\"hour\"),\n                ((pl.col(\"total_seconds\") % 3600) // 60).alias(\"minute\"),\n                (pl.col(\"total_seconds\") % 60).alias(\"second\"),\n            ]\n        )\n        .collect()\n        .to_pandas()\n    )\n\n    df = feat_eng(df)\n    \n    return df\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:06:14.700831Z","iopub.execute_input":"2024-11-23T19:06:14.702212Z","iopub.status.idle":"2024-11-23T19:06:14.708973Z","shell.execute_reply.started":"2024-11-23T19:06:14.702168Z","shell.execute_reply":"2024-11-23T19:06:14.707761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm.auto import tqdm \nfrom joblib import Parallel, delayed\nfrom time import sleep, time\nfrom multiprocessing import cpu_count\nimport gc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T18:52:07.534644Z","iopub.execute_input":"2024-11-23T18:52:07.535080Z","iopub.status.idle":"2024-11-23T18:52:07.540256Z","shell.execute_reply.started":"2024-11-23T18:52:07.535044Z","shell.execute_reply":"2024-11-23T18:52:07.539105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nworking_folder = Path(\"/kaggle/working/\")\nimages_folder = working_folder/\"imagesandannotations\"\nimages_folder.mkdir()\n\nwindow_properties_batch_file = os.path.join(images_folder, \"window_properties.json\")\nall_events_batch_file = os.path.join(images_folder, \"all_events.json\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T18:59:00.063051Z","iopub.execute_input":"2024-11-23T18:59:00.063455Z","iopub.status.idle":"2024-11-23T18:59:00.190165Z","shell.execute_reply.started":"2024-11-23T18:59:00.063420Z","shell.execute_reply":"2024-11-23T18:59:00.188760Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_ids = df['id'].unique()\nlen(series_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T18:58:55.049526Z","iopub.execute_input":"2024-11-23T18:58:55.050404Z","iopub.status.idle":"2024-11-23T18:58:55.057746Z","shell.execute_reply.started":"2024-11-23T18:58:55.050361Z","shell.execute_reply":"2024-11-23T18:58:55.056440Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_ids = series_ids[0:190]\n#series_ids = series_ids[0:2]\nseries_ids","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T18:58:56.508800Z","iopub.execute_input":"2024-11-23T18:58:56.509182Z","iopub.status.idle":"2024-11-23T18:58:56.516094Z","shell.execute_reply.started":"2024-11-23T18:58:56.509149Z","shell.execute_reply":"2024-11-23T18:58:56.515001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport gc\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom tqdm import tqdm\nimport numpy as np  # Ensure np is imported for handling int64 types\n\ndef process_and_plot(series_ids, images_folder):\n    \n    # Initialize main lists to hold all data\n    window_properties = []\n    all_events = []\n    \n    # Batch tracking lists\n    batch_size = 50  # Adjust batch size to your memory constraints\n    batch_window_properties = []\n    batch_all_events = []\n    \n    # Files for saving batches of data\n    window_properties_batch_file = images_folder / \"window_properties.json\"\n    all_events_batch_file = images_folder / \"all_events.json\"\n\n    # Create files if they don't exist\n    window_properties_batch_file.touch(exist_ok=True)\n    all_events_batch_file.touch(exist_ok=True)\n\n    # Process the series_ids\n    for idx in tqdm(series_ids):\n        \n        filtered_data = feat_eng_by_id(idx)\n\n        series = filtered_data.reset_index(drop=True)\n        series['color'] = [\"blue\" if large_enmo else \"green\" for large_enmo in series['large_enmo']]\n        series['timestamp'] = pd.to_datetime(series['timestamp'])\n        series['timestamp'] = series['timestamp'].apply(lambda x: x if x.tzinfo is not None else x.tz_localize('UTC'))\n        series['timestamp_utc'] = series['timestamp'].map(lambda timestamp: timestamp.astimezone(timezone.utc))\n        series['anglez_radians'] = (np.pi / 180) * series['anglez']\n        series['cos_anglez'] = np.cos(series['anglez_radians'])\n        series['enmo'] = np.clip(series['enmo'], 0, 1)\n        min_date_utc = series['timestamp_utc'].dt.date.min()\n        max_date_utc = series['timestamp_utc'].dt.date.max()\n        \n        series_24_hour_windows = {}\n        upper_bound = datetime(year=min_date_utc.year, month=min_date_utc.month, day=min_date_utc.day, hour=20, minute=30, tzinfo=timezone.utc)\n        lower_bound = upper_bound + timedelta(hours=-24) # 8:30pm UTC on the previous day.\n        while lower_bound < series['timestamp_utc'].max():\n            window_df = series.loc[(series['timestamp_utc'] >= lower_bound) & (series['timestamp_utc'] < upper_bound)].reset_index(drop=True)\n            if len(window_df) > 0:\n                series_24_hour_windows[upper_bound.isoformat()[:-6]] = window_df\n            upper_bound += timedelta(hours=24)\n            lower_bound += timedelta(hours=24)\n        \n        windows = list(series_24_hour_windows.keys())\n        num_steps_cumulative = 0\n        \n        for window_idx, window in enumerate(windows):\n            \n             if (series_24_hour_windows[window]['non-wear_flag'].mean()<0.5) & (len(series_24_hour_windows[window]) == 17280): \n\n                day = series_24_hour_windows[window]['which_day'].iloc[0]\n                 \n                fig = plt.figure(figsize=(14.4, 4))  # (width, height) in inches\n                #plt.plot(series_24_hour_windows[window]['timestamp_utc'], series_24_hour_windows[window]['cos_anglez'], color=\"red\")\n                plt.plot(series_24_hour_windows[window]['timestamp_utc'],\n                             series_24_hour_windows[window]['anglezdiffabs_median_60_norm'],\n                             color=\"red\")\n                plt.scatter(\n                        series_24_hour_windows[window]['timestamp_utc'], \n                        series_24_hour_windows[window]['enmo'], \n                        color=series_24_hour_windows[window]['color'], \n                        s=1\n                    )\n                plt.scatter(series_24_hour_windows[window]['timestamp_utc'], series_24_hour_windows[window]['non-wear_flag'], label='non_wear_flag', color='red', alpha=0.7, s=10)\n        \n                plt.fill_between(series_24_hour_windows[window]['timestamp_utc'],\n                                 0, max(1,series_24_hour_windows[window]['anglezdiffabs_median_60_norm'].max()), \n                                 where=(series_24_hour_windows[window]['non-wear_flag'] == 1), \n                                 color='red', alpha=0.1, label='Day Period')\n                 \n                plt.fill_between(series_24_hour_windows[window]['timestamp_utc'],\n                                 0, max(1,series_24_hour_windows[window]['anglezdiffabs_median_60_norm'].max()), \n                                 where=(series_24_hour_windows[window]['day_period_b'] == 1), \n                                 color='blue', alpha=0.1, label='Day Period')\n                ax = plt.gca()\n                ax.spines['top'].set_visible(False)\n                ax.spines['right'].set_visible(False)\n                ax.spines['bottom'].set_visible(False)\n                ax.spines['left'].set_visible(False)\n                ax.set_xticks([])\n                ax.set_yticks([])\n                plt.margins(0, 0)\n                plt.subplots_adjust(left=0, right=1, top=1, bottom=0)\n        \n                # Save image to images_folder\n                image_path = images_folder / f\"{idx}_{day}.jpg\"\n                plt.savefig(image_path, dpi=150, bbox_inches=\"tight\", pad_inches=0)\n                plt.close(fig)\n\n                # Store data for batch, safely converting where necessary\n                window_properties_item = {\n                        'series_id': str(idx),  # Ensure series_id is stored as string\n                        'image_name': f\"{idx}_{day}.jpg\", \n                        'idx_in_series': int(day)  # Convert 'day' to int if it's a number\n                    }\n                window_properties.append(window_properties_item)\n                batch_window_properties.append(window_properties_item)  # Add to batch\n\n                # Add to all_events batch\n                sii_label = df.loc[df['id'] == idx, 'sii'].values.item()\n\n                # Ensure sii_label is numeric before converting to int\n                if isinstance(sii_label, (int, np.integer)):\n                    sii_label = int(sii_label)\n                else:\n                    sii_label = str(sii_label)  # If it's a string, store it as string\n\n                all_events_item = {'series_id': str(idx), 'image_name': f\"{idx}_{day}.jpg\", 'label': sii_label}\n                all_events.append(all_events_item)\n                batch_all_events.append(all_events_item)  # Add to batch\n\n                # Save in batches when batch size is reached\n                if len(batch_window_properties) >= batch_size:\n                    # Save window_properties batch to JSON file\n                    with open(window_properties_batch_file, \"a\") as f:\n                        json.dump(batch_window_properties, f, indent=4)\n                    batch_window_properties.clear()  # Clear the batch\n\n                    # Save all_events batch to JSON file\n                    with open(all_events_batch_file, \"a\") as f:\n                        json.dump(batch_all_events, f, indent=4)\n                    batch_all_events.clear()  # Clear the batch\n\n                gc.collect()  # Clean up memory after each batch\n\n    # After the loop, save any remaining data if the final batch wasn't full\n    if batch_window_properties:\n        with open(window_properties_batch_file, \"a\") as f:\n            json.dump(batch_window_properties, f, indent=4)\n    if batch_all_events:\n        with open(all_events_batch_file, \"a\") as f:\n            json.dump(batch_all_events, f, indent=4)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:31:02.197998Z","iopub.execute_input":"2024-11-23T19:31:02.198398Z","iopub.status.idle":"2024-11-23T19:31:02.225619Z","shell.execute_reply.started":"2024-11-23T19:31:02.198365Z","shell.execute_reply":"2024-11-23T19:31:02.224453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"process_and_plot(series_ids,images_folder)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:31:06.240833Z","iopub.execute_input":"2024-11-23T19:31:06.241352Z","iopub.status.idle":"2024-11-23T19:31:31.744102Z","shell.execute_reply.started":"2024-11-23T19:31:06.241317Z","shell.execute_reply":"2024-11-23T19:31:31.742963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:32:54.192938Z","iopub.execute_input":"2024-11-23T19:32:54.194039Z","iopub.status.idle":"2024-11-23T19:32:54.198701Z","shell.execute_reply.started":"2024-11-23T19:32:54.193995Z","shell.execute_reply":"2024-11-23T19:32:54.197581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_all_data_from_batches(directory_path, file_prefix):\n    all_data = []\n    \n    # Get all filenames matching the prefix and pattern\n    batch_files = glob.glob(f\"{directory_path}/{file_prefix}*.json\")\n    \n    for batch_file in batch_files:\n        with open(batch_file, 'r') as f:\n            try:\n                data = json.load(f)\n                all_data.extend(data)  # Append data from this batch to the overall list\n            except json.JSONDecodeError as e:\n                print(f\"Error loading {batch_file}: {e}\")\n    \n    return all_data\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:32:55.806765Z","iopub.execute_input":"2024-11-23T19:32:55.808096Z","iopub.status.idle":"2024-11-23T19:32:55.814154Z","shell.execute_reply.started":"2024-11-23T19:32:55.808052Z","shell.execute_reply":"2024-11-23T19:32:55.813099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Path to your JSON files\nwindow_properties_path = '/kaggle/working/imagesandannotations/window_properties.json'\nall_events_path = '/kaggle/working/imagesandannotations/all_events.json'\n\n# Function to load and fix the JSON files if needed\ndef load_json_file(file_path):\n    with open(file_path, 'r') as file:\n        content = file.read()\n\n    # Remove any invalid '][' and ensure the content is wrapped in square brackets\n    fixed_content = content.replace(\"][\", \",\")  # Fix '][' by replacing with a comma\n\n    # Wrap content in square brackets if it's not already\n    if not fixed_content.startswith('['):\n        fixed_content = f\"[{fixed_content}\"\n    if not fixed_content.endswith(']'):\n        fixed_content = f\"{fixed_content}]\"\n\n    # Try to load the corrected content\n    try:\n        data = json.loads(fixed_content)\n        print(f\"Successfully loaded {file_path}!\")\n        return data\n    except json.JSONDecodeError as e:\n        print(f\"Error decoding JSON in {file_path}: {e}\")\n        return None\n\n# Load both files\nwindow_properties_data = load_json_file(window_properties_path)\nall_events_data = load_json_file(all_events_path)\n\n# Print a sample from each to confirm the data is loaded correctly\nprint(\"Sample from window_properties:\")\nprint(window_properties_data[:3])  # Show the first 3 items as a sample\n\nprint(\"\\nSample from all_events:\")\nprint(all_events_data[:3])  # Show the first 3 items as a sample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:34:22.443933Z","iopub.execute_input":"2024-11-23T19:34:22.444351Z","iopub.status.idle":"2024-11-23T19:34:22.454773Z","shell.execute_reply.started":"2024-11-23T19:34:22.444314Z","shell.execute_reply":"2024-11-23T19:34:22.453355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"window_properties_df = pd.DataFrame(window_properties_data)\nwindow_properties_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:34:41.622570Z","iopub.execute_input":"2024-11-23T19:34:41.622987Z","iopub.status.idle":"2024-11-23T19:34:41.634361Z","shell.execute_reply.started":"2024-11-23T19:34:41.622952Z","shell.execute_reply":"2024-11-23T19:34:41.633151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"annotations_df = pd.DataFrame(all_events_data)\nannotations_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:34:43.290146Z","iopub.execute_input":"2024-11-23T19:34:43.290540Z","iopub.status.idle":"2024-11-23T19:34:43.302431Z","shell.execute_reply.started":"2024-11-23T19:34:43.290505Z","shell.execute_reply":"2024-11-23T19:34:43.301144Z"}},"outputs":[],"execution_count":null}]}