{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":38760,"databundleVersionId":4493939,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import cudf\ncudf.__version__","metadata":{"execution":{"iopub.status.busy":"2024-09-09T14:48:48.454528Z","iopub.execute_input":"2024-09-09T14:48:48.454886Z","iopub.status.idle":"2024-09-09T14:48:55.123343Z","shell.execute_reply.started":"2024-09-09T14:48:48.454846Z","shell.execute_reply":"2024-09-09T14:48:55.122366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df = cudf.read_json('/kaggle/input/otto-recommender-system/test.jsonl')","metadata":{"execution":{"iopub.status.busy":"2024-09-09T14:49:05.721623Z","iopub.execute_input":"2024-09-09T14:49:05.72327Z","iopub.status.idle":"2024-09-09T14:49:11.991095Z","shell.execute_reply.started":"2024-09-09T14:49:05.72321Z","shell.execute_reply":"2024-09-09T14:49:11.989632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cudf\nimport pandas as pd\nimport warnings\nwarnings.filterwarnings(action='ignore')\n\nlines = []  # List to hold all valid lines in the file\nwith open('/kaggle/input/otto-recommender-system/test.jsonl', 'r') as f:\n    for line in f:\n        try:\n            lines.append(pd.read_json(line)) # Try reading each line as a separate JSON object\n        except ValueError:  # If it fails to read the line skip it and continue with the next one\n            continue\n\ndf = cudf.from_pandas(pd.concat(lines, axis=0))  # Convert all valid lines into a single DataFrame","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-07T13:42:17.784164Z","iopub.execute_input":"2024-09-07T13:42:17.784503Z","iopub.status.idle":"2024-09-07T14:49:05.090162Z","shell.execute_reply.started":"2024-09-07T13:42:17.784462Z","shell.execute_reply":"2024-09-07T14:49:05.089229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_parquet(\"test.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-09-07T14:56:56.053892Z","iopub.execute_input":"2024-09-07T14:56:56.054836Z","iopub.status.idle":"2024-09-07T14:56:56.951076Z","shell.execute_reply.started":"2024-09-07T14:56:56.054785Z","shell.execute_reply":"2024-09-07T14:56:56.949977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"import cudf\nimport pandas as pd\nimport warnings\nwarnings.filterwarnings(action='ignore')\n\nlines = []  # List to hold all valid lines in the file\nwith open('/kaggle/input/otto-recommender-system/train.jsonl', 'r') as f:\n    for line in f:\n        try:\n            lines.append(pd.read_json(line)) # Try reading each line as a separate JSON object\n        except ValueError:  # If it fails to read the line skip it and continue with the next one\n            continue\n\ntrain_df = cudf.from_pandas(pd.concat(lines, axis=0))  # Convert all valid lines into a single DataFrame","metadata":{"execution":{"iopub.status.busy":"2024-09-09T11:29:33.41696Z","iopub.execute_input":"2024-09-09T11:29:33.417816Z","iopub.status.idle":"2024-09-09T11:29:53.606006Z","shell.execute_reply.started":"2024-09-09T11:29:33.417769Z","shell.execute_reply":"2024-09-09T11:29:53.604724Z"}}},{"cell_type":"code","source":"#train_df.to_parquet(\"test.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-09-09T11:29:53.606668Z","iopub.status.idle":"2024-09-09T11:29:53.607026Z","shell.execute_reply.started":"2024-09-09T11:29:53.606837Z","shell.execute_reply":"2024-09-09T11:29:53.606854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cudf\nimport pandas as pd\nimport warnings\nwarnings.filterwarnings(action='ignore')\n\nchunks = []  # List to hold chunks of dataframe\nlines = []  # List to hold each line in current chunk\nwith open('/kaggle/input/otto-recommender-system/train.jsonl', 'r') as f:\n    for line in f:\n        try:\n            lines.append(pd.read_json(line)) # Try reading the line as a separate JSON object\n        except ValueError:  # If it fails to read the line skip it and continue with the next one\n            continue\n        else:\n            if len(lines) == 750000:  # When we have 1000 valid lines\n                df = cudf.from_pandas(pd.concat(lines, axis=0)) # Create a dataframe from them\n                chunks.append(df)\n                lines.clear()  # Clear the list for next set of lines\n    if lines:  # If there are remaining valid lines after the loop ends\n        df = cudf.from_pandas(pd.concat(lines, axis=0)) # Create a dataframe from them and append to chunks\n        chunks.append(df)\n\nfinal_df = pd.concat(chunks, axis=0)  # Finally concatenate all chunks together into one DataFrame","metadata":{"execution":{"iopub.status.busy":"2024-09-09T14:49:52.082263Z","iopub.execute_input":"2024-09-09T14:49:52.082686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.to_parquet(\"train.parquet\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}