{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom tqdm import tqdm\n","metadata":{"execution":{"iopub.status.busy":"2022-11-04T08:31:07.335235Z","iopub.execute_input":"2022-11-04T08:31:07.336602Z","iopub.status.idle":"2022-11-04T08:31:07.360363Z","shell.execute_reply.started":"2022-11-04T08:31:07.336537Z","shell.execute_reply":"2022-11-04T08:31:07.359192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport json\n#you need to add you path here\nwith open(os.path.join('../input/otto-recommender-system/test.jsonl'), 'r',\n          encoding='utf-8') as f1:\n    ll = [json.loads(line.strip()) for line in f1.readlines()]\n\n    #this is the total length size of the json file\n    print(len(ll))\n\n    #in here 2000 means we getting splits of 2000 tweets\n    #you can define your own size of split according to your need\n    size_of_the_split=2000\n    total = len(ll) // size_of_the_split\n\n    #in here you will get the Number of splits\n    print(total+1)\n\n    for i in range(total+1):\n        json.dump(ll[i * size_of_the_split:(i + 1) * size_of_the_split], open(\n            \"./\" + str(i+1) + \".json\", 'w',\n            encoding='utf8'), ensure_ascii=False, indent=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-04T09:14:36.398516Z","iopub.execute_input":"2022-11-04T09:14:36.399034Z","iopub.status.idle":"2022-11-04T09:16:27.920888Z","shell.execute_reply.started":"2022-11-04T09:14:36.398928Z","shell.execute_reply":"2022-11-04T09:16:27.919508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n#in here you need to provide the number spilts\nnumber_of_splits=7\nfor i in range(0,number_of_splits+1):\n    \n    word =i+1\n    print(word)\n\n    json_file =f\"{word}.json\"\n    csv_file =f\"{word}.csv\"\n    df = pd.read_json(fr'./{json_file}')\n    df.to_csv (fr'./{csv_file}', index = None)","metadata":{"execution":{"iopub.status.busy":"2022-11-04T09:31:51.889146Z","iopub.execute_input":"2022-11-04T09:31:51.889644Z","iopub.status.idle":"2022-11-04T09:31:52.327293Z","shell.execute_reply.started":"2022-11-04T09:31:51.889606Z","shell.execute_reply":"2022-11-04T09:31:52.326451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df =pd.read_csv('./2.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-04T09:29:22.705930Z","iopub.execute_input":"2022-11-04T09:29:22.706704Z","iopub.status.idle":"2022-11-04T09:29:22.727753Z","shell.execute_reply.started":"2022-11-04T09:29:22.706639Z","shell.execute_reply":"2022-11-04T09:29:22.726894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-04T09:29:29.695953Z","iopub.execute_input":"2022-11-04T09:29:29.696349Z","iopub.status.idle":"2022-11-04T09:29:29.715586Z","shell.execute_reply.started":"2022-11-04T09:29:29.696317Z","shell.execute_reply":"2022-11-04T09:29:29.714205Z"},"trusted":true},"execution_count":null,"outputs":[]}]}