{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-17T23:52:53.291745Z","iopub.execute_input":"2024-10-17T23:52:53.292137Z","iopub.status.idle":"2024-10-17T23:52:53.880699Z","shell.execute_reply.started":"2024-10-17T23:52:53.292094Z","shell.execute_reply":"2024-10-17T23:52:53.879313Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install polars[gpu]","metadata":{"execution":{"iopub.status.busy":"2024-10-17T23:52:53.88258Z","iopub.execute_input":"2024-10-17T23:52:53.883429Z","iopub.status.idle":"2024-10-17T23:53:14.777058Z","shell.execute_reply.started":"2024-10-17T23:52:53.883365Z","shell.execute_reply":"2024-10-17T23:53:14.771676Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\nimport lightgbm as lgb\nimport pandas as pd\nimport numpy as np\nfrom pprint import pprint","metadata":{"execution":{"iopub.status.busy":"2024-10-17T23:53:14.783846Z","iopub.execute_input":"2024-10-17T23:53:14.785209Z","iopub.status.idle":"2024-10-17T23:53:28.681879Z","shell.execute_reply.started":"2024-10-17T23:53:14.784989Z","shell.execute_reply":"2024-10-17T23:53:28.677535Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##値の指定\ntarget = \"responder_6\"\nstate = 42\ndev_strt_id = 4_500_000\nversion_nb = \"V1_1\"\noffline_strt_dt = 500","metadata":{"execution":{"iopub.status.busy":"2024-10-17T23:53:28.688438Z","iopub.execute_input":"2024-10-17T23:53:28.690635Z","iopub.status.idle":"2024-10-17T23:53:28.711827Z","shell.execute_reply.started":"2024-10-17T23:53:28.690536Z","shell.execute_reply":"2024-10-17T23:53:28.706411Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pl.scan_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet').\\\n        select(\n            pl.int_range(pl.len(), dtype=pl.UInt32).alias(\"id\"),\n            pl.all(),)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T23:53:28.716397Z","iopub.execute_input":"2024-10-17T23:53:28.717695Z","iopub.status.idle":"2024-10-17T23:53:28.773358Z","shell.execute_reply.started":"2024-10-17T23:53:28.71753Z","shell.execute_reply":"2024-10-17T23:53:28.768311Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_cols = train_df.collect_schema().names()\nsel_cols = [c for c in all_cols\n           if c.startswith((\"responder\", \"weight\", \"id\", \"date_id\", \"time_id\", \"partition_id\"))\n                == False]\ntargets = [c for c in all_cols if(c.startswith(\"resopnder\") == True)]\n\nsample_weight = train_df.select(pl.col(\"weight\")).collect().to_series()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T23:53:28.777861Z","iopub.execute_input":"2024-10-17T23:53:28.779746Z","iopub.status.idle":"2024-10-17T23:53:29.321553Z","shell.execute_reply.started":"2024-10-17T23:53:28.779667Z","shell.execute_reply":"2024-10-17T23:53:29.318123Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len_train = train_df.select(pl.col(\"date_id\")).collect().shape[0]\nlen_ofl_mdl = len_train - dev_strt_id\nlast_tr_dt = train_df.select(pl.col(\"date_id\")).collect().row(len_ofl_mdl)[0]\n\nXYtrain = train_df.filter(pl.col(\"date_id\").le(last_tr_dt))\nXYdev = train_df.filter(pl.col(\"date_id\").gt(last_tr_dt))\nXYdev.collect().write_parquet(\"XYdev\", partition_by=\"date_id\")\nXYtrain.collect().write_parquet(\"XYtrain\", partition_by=\"date_id\")","metadata":{"execution":{"iopub.status.busy":"2024-10-17T23:53:31.609487Z","iopub.execute_input":"2024-10-17T23:53:31.613426Z","iopub.status.idle":"2024-10-17T23:54:21.698423Z","shell.execute_reply.started":"2024-10-17T23:53:31.613247Z","shell.execute_reply":"2024-10-17T23:54:21.695997Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os \npath ='/kaggle/working/'\nfileList = os.listdir(path)\nfor f in fileList:\n    print(f)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T23:55:06.575604Z","iopub.execute_input":"2024-10-17T23:55:06.577162Z","iopub.status.idle":"2024-10-17T23:55:06.587169Z","shell.execute_reply.started":"2024-10-17T23:55:06.577101Z","shell.execute_reply":"2024-10-17T23:55:06.584503Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = XYtrain\nsel_cols = [c for c in sel_cols if c not in [\"id\", \"date_id\", \"time_id\", \"partition_id\"]]\ndrop_cols = [c for c in sel_cols if \"responder\" in c] + [\"weight\"]\nXtrain = train.filter(pl.col(\"date_id\").gt(offline_strt_dt)).\\\n        select(pl.col(sel_cols)).\\\n        collect(engine=\"gpu\").to_pandas()\n\nXtrain.index = range(len(Xtrain))\nytrain = Xtrain[target]\nsw_tr = Xtrain[\"weight\"].values.flatten()\nXtrain = Xtrain.drop(drop_cols, axis=1, errors = \"ignore\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from Ipython.display import FileLinks\n","metadata":{},"outputs":[],"execution_count":null}]}