{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%writefile baseline.py\nimport polars as pl#similar to pandas, but with better performance when dealing with large datasets.\nimport pandas as pd#read csv,parquet\nimport numpy as np#for scientific computation of matrices\nimport gc#rubbish collection\nimport os#interact with operation system\n#environment provided by competition hoster\nimport kaggle_evaluation.jane_street_inference_server\nimport warnings#avoid some negligible errors\n#The filterwarnings () method is used to set warning filters, which can control the output method and level of warning information.\nwarnings.filterwarnings('ignore')\n\nfor notebookid in range(3):\n    print(f\"notebookid:{notebookid}\")\n\n    all_datas=[]\n    dataid=[[4,5,6,7],[5,6,7,8],[6,7,8,9]]\n    train_times=[280,275,275]\n    test_times=[125,125,125]\n    for idx in dataid[notebookid]:\n        train=pl.read_parquet(f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={idx}/part-0.parquet\")\n        train=train.to_pandas()\n        print(f\"train.shape:{train.shape}\")\n        all_datas.append(train)\n    all_datas=pd.concat(all_datas)\n    all_datas=all_datas[-15000000:]\n    print(f\"len(all_datas):{len(all_datas)}\")\n    print(f\"date_id_min:{all_datas['date_id'].min()},date_id_max:{all_datas['date_id'].max()}\")\n\n    train_cols=['date_id', 'time_id', 'symbol_id', 'weight', 'feature_00', 'feature_01', 'feature_02', 'feature_03', 'feature_04', 'feature_05', 'feature_06', 'feature_07', 'feature_08', 'feature_09', 'feature_10', 'feature_11', 'feature_12', 'feature_13', 'feature_14', 'feature_15', 'feature_16', 'feature_17', 'feature_18', 'feature_19', 'feature_20', 'feature_21', 'feature_22', 'feature_23', 'feature_24', 'feature_25', 'feature_26', 'feature_27', 'feature_28', 'feature_29', 'feature_30', 'feature_31', 'feature_32', 'feature_33', 'feature_34', 'feature_35', 'feature_36', 'feature_37', 'feature_38', 'feature_39', 'feature_40', 'feature_41', 'feature_42', 'feature_43', 'feature_44', 'feature_45', 'feature_46', 'feature_47', 'feature_48', 'feature_49', 'feature_50', 'feature_51', 'feature_52', 'feature_53', 'feature_54', 'feature_55', 'feature_56', 'feature_57', 'feature_58', 'feature_59', 'feature_60', 'feature_61', 'feature_62', 'feature_63', 'feature_64', 'feature_65', 'feature_66', 'feature_67', 'feature_68', 'feature_69', 'feature_70', 'feature_71', 'feature_72', 'feature_73', 'feature_74', 'feature_75', 'feature_76', 'feature_77', 'feature_78', 'responder_0', 'responder_1', 'responder_2', 'responder_3', 'responder_4', 'responder_5', 'responder_6', 'responder_7', 'responder_8']\n    lags_cols=['date_id', 'time_id','symbol_id', 'responder_0_lag_1', 'responder_1_lag_1', 'responder_2_lag_1', 'responder_3_lag_1', 'responder_4_lag_1', 'responder_5_lag_1', 'responder_6_lag_1', 'responder_7_lag_1', 'responder_8_lag_1']\n    \n    train=all_datas\n    del all_datas\n    gc.collect()\n    \n    print(\"< train_lags.parquet >\")\n    for i in range(9):\n        train[f'responder_{i}_lag_1']=train[f'responder_{i}']\n    train_lags=train[lags_cols].reset_index(drop=True)\n    train_lags['date_id']=train_lags['date_id']-train_lags['date_id'].min()+1\n    train_lags=train_lags[train_lags['date_id']>=1]\n\n    print(\"< train.parquet >\")\n    train=train[train_cols].reset_index(drop=True)\n    train['date_id']=train['date_id']-train['date_id'].min()\n    train=train[train['date_id']>=1]\n    \n    print(f\"len(train):{len(train)},len(train_lags):{len(train_lags)}\")\n    \n    \n    os.makedirs('synthetic_train.parquet', exist_ok=True) \n    os.makedirs('synthetic_valid.parquet', exist_ok=True) \n\n    num_folds=1\n    gap_time=15#half of the month\n    train_time=train_times[notebookid]\n    \n    for fold in range(num_folds):\n        \n        os.makedirs(f'synthetic_train.parquet/fold={fold+notebookid}/', exist_ok=True) \n        os.makedirs(f'synthetic_train_lags.parquet/fold={fold+notebookid}/', exist_ok=True) \n        os.makedirs(f'synthetic_valid.parquet/fold={fold+notebookid}/', exist_ok=True) \n        os.makedirs(f'synthetic_valid_lags.parquet/fold={fold+notebookid}/', exist_ok=True)\n        \n        train_start_time=train['date_id'].min()\n        train_end_time=train_start_time+train_time\n        valid_start_time=train_end_time+gap_time\n        valid_end_time=valid_start_time+test_times[notebookid]#around 450 test datas\n        print(f\"train_start_time:{train_start_time},train_end_time:{train_end_time}\")\n        print(f\"valid_start_time:{valid_start_time},valid_end_time:{valid_end_time}\")\n    \n        print(f\"< fold{fold+1} train data >\")\n        train_fold=train[(train['date_id']>=train_start_time)&(train['date_id']<train_end_time)].copy()\n        train_lags_fold=train_lags[(train_lags['date_id']>=train_start_time)&(train_lags['date_id']<train_end_time)].copy()\n        train_fold.to_parquet(f'synthetic_train.parquet/fold={fold+notebookid}/part-0.parquet',index=None)\n        train_lags_fold.to_parquet(f'synthetic_train_lags.parquet/fold={fold+notebookid}/part-0.parquet',index=None)\n        \n        print(f\"< fold{fold+1} valid data >\")\n        valid_fold=train[(train['date_id']>=valid_start_time)&(train['date_id']<valid_end_time)].copy()\n        valid_lags_fold=train_lags[(train_lags['date_id']>=valid_start_time)&(train_lags['date_id']<valid_end_time)].copy()\n        valid_fold.to_parquet(f'synthetic_valid.parquet/fold={fold+notebookid}/part-0.parquet',index=None)\n        valid_lags_fold.to_parquet(f'synthetic_valid_lags.parquet/fold={fold+notebookid}/part-0.parquet',index=None)\n        \n        print(f\"train_fold.shape:{train_fold.shape},train_lags_fold.shape:{train_lags_fold.shape}\")\n        print(f\"valid_fold.shape:{valid_fold.shape},valid_lags_fold.shape:{valid_lags_fold.shape}\")\n        print(\"-\"*30)\n        del train_fold,train_lags_fold,valid_fold,valid_lags_fold\n        gc.collect()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python baseline.py","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}