{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# AI4C_Data_Preprocessing\n\nA step by step approach to data preprocessing, rather than all functions. My aim is to well understand data preparations using only pandas.\nCode is similar to the ones already available in kaggle competition AI4C. My version is just a little bit different.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport zipfile\nimport os\nfrom pathlib import Path\nimport random\nfrom tqdm import tqdm\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.width = 180\npd.options.display.max_colwidth = 120\npd.options.mode.chained_assignment = None\npd.options.display.float_format = '{:.4f}'.format\n\nfrom sklearn.model_selection import GroupShuffleSplit\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:05:16.020311Z","iopub.execute_input":"2022-07-19T12:05:16.021247Z","iopub.status.idle":"2022-07-19T12:05:22.540306Z","shell.execute_reply.started":"2022-07-19T12:05:16.021209Z","shell.execute_reply":"2022-07-19T12:05:22.539292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# some details\nprint(f\"Numbers of json files in train set:\", len(os.listdir(\"/kaggle/input/AI4Code/train\")))\nprint(f\"Numbers of json files in trest set:\", len(os.listdir(\"/kaggle/input/AI4Code/test\")))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:26:55.738062Z","iopub.execute_input":"2022-07-19T10:26:55.738418Z","iopub.status.idle":"2022-07-19T10:26:55.819806Z","shell.execute_reply.started":"2022-07-19T10:26:55.738387Z","shell.execute_reply":"2022-07-19T10:26:55.818684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Train and Test\ntrain_dir = Path(\"/kaggle/input/AI4Code/train\")\ntest_dir = Path(\"/kaggle/input/AI4Code/test\")\n\ndef read_nb(path):\n  return(pd.read_json(path, dtype={'cell_type': 'category', 'source': 'str'}).assign(id = path.stem)).rename_axis('cell_id')\n  \npaths_train = list((train_dir).glob('*.json'))\npaths_test = list((test_dir).glob('*.json'))\n\nnb_train = [read_nb(path) for path in paths_train]\nnb_test = [read_nb(path) for path in paths_test]","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:29:52.111323Z","iopub.execute_input":"2022-07-19T10:29:52.112036Z","iopub.status.idle":"2022-07-19T10:47:46.672048Z","shell.execute_reply.started":"2022-07-19T10:29:52.111999Z","shell.execute_reply":"2022-07-19T10:47:46.671060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make Datasets","metadata":{}},{"cell_type":"markdown","source":"### Train","metadata":{}},{"cell_type":"code","source":"df = pd.concat(nb_train).set_index('id', append=True).swaplevel().sort_index(level='id', sort_remaining=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:14:48.245836Z","iopub.execute_input":"2022-07-19T11:14:48.246188Z","iopub.status.idle":"2022-07-19T11:15:40.436054Z","shell.execute_reply.started":"2022-07-19T11:14:48.246159Z","shell.execute_reply":"2022-07-19T11:15:40.434985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:18:54.369952Z","iopub.execute_input":"2022-07-19T11:18:54.370320Z","iopub.status.idle":"2022-07-19T11:18:54.387641Z","shell.execute_reply.started":"2022-07-19T11:18:54.370286Z","shell.execute_reply":"2022-07-19T11:18:54.386648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test","metadata":{}},{"cell_type":"code","source":"test_df = pd.concat(nb_test).set_index('id', append=True).swaplevel().sort_index(level='id', sort_remaining=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:00:02.210169Z","iopub.execute_input":"2022-07-19T11:00:02.210520Z","iopub.status.idle":"2022-07-19T11:00:02.221495Z","shell.execute_reply.started":"2022-07-19T11:00:02.210486Z","shell.execute_reply":"2022-07-19T11:00:02.220446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.to_csv('test_df.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:00:09.997904Z","iopub.execute_input":"2022-07-19T11:00:09.998466Z","iopub.status.idle":"2022-07-19T11:00:10.011015Z","shell.execute_reply.started":"2022-07-19T11:00:09.998431Z","shell.execute_reply":"2022-07-19T11:00:10.009807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Ranked and Ancestors","metadata":{}},{"cell_type":"code","source":"# Ordered\ndf_orders = pd.read_csv(\"/kaggle/input/AI4Code/train_orders.csv\", index_col = \"id\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:00:24.258370Z","iopub.execute_input":"2022-07-19T11:00:24.258834Z","iopub.status.idle":"2022-07-19T11:00:26.010966Z","shell.execute_reply.started":"2022-07-19T11:00:24.258802Z","shell.execute_reply":"2022-07-19T11:00:26.009892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_orders['cell_order'] = df_orders['cell_order'].apply(str.split) # split cell_order ids\n\ndef get_ranks(ordered, messed):\n    return [ordered.index(d) for d in messed]\n\ndf_cells = df_orders.join(df.reset_index('cell_id').groupby('id')['cell_id'].apply(list), how='right')\n\nranks = {}\n\nfor id_, cell_order, cell_id in df_cells.itertuples():\n    ranks[id_] = {'cell_id': cell_id, 'rank': get_ranks(cell_order, cell_id)}\n\ndf_ranks = (pd.DataFrame.from_dict(ranks, orient='index').rename_axis('id').apply(pd.Series.explode).set_index('cell_id', append=True))\n\ndf_ranks","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:00:27.783438Z","iopub.execute_input":"2022-07-19T11:00:27.784923Z","iopub.status.idle":"2022-07-19T11:01:21.694485Z","shell.execute_reply.started":"2022-07-19T11:00:27.784877Z","shell.execute_reply":"2022-07-19T11:01:21.693505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ancestors\ndf_ancestors = pd.read_csv('/kaggle/input/AI4Code/train_ancestors.csv', index_col='id')\ndf_ancestors","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:01:43.509715Z","iopub.execute_input":"2022-07-19T11:01:43.510297Z","iopub.status.idle":"2022-07-19T11:01:43.713546Z","shell.execute_reply.started":"2022-07-19T11:01:43.510258Z","shell.execute_reply":"2022-07-19T11:01:43.712561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#adding rankings and ancestors\ndf.reset_index(drop=False, inplace=True)\ndf = df.merge(df_ranks, on = ['id','cell_id'])\ndf = df.merge(df_ancestors, on=[\"id\"])\n\n# add percentage rank (labels) and sort\ndf[\"pct_rank\"] = df[\"rank\"] / df.groupby(\"id\")[\"cell_id\"].transform(\"count\")\ndf = df.sort_values(\"pct_rank\").reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:19:11.070463Z","iopub.execute_input":"2022-07-19T11:19:11.071153Z","iopub.status.idle":"2022-07-19T11:20:00.749075Z","shell.execute_reply.started":"2022-07-19T11:19:11.071115Z","shell.execute_reply":"2022-07-19T11:20:00.746480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:20:52.089211Z","iopub.execute_input":"2022-07-19T11:20:52.089599Z","iopub.status.idle":"2022-07-19T11:20:52.108422Z","shell.execute_reply.started":"2022-07-19T11:20:52.089569Z","shell.execute_reply":"2022-07-19T11:20:52.107437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## MARKDOWN and CODE datasets","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import GroupShuffleSplit\n\nNVALID = 0.1  # size of validation set\n\nsplitter = GroupShuffleSplit(n_splits=1, test_size=NVALID, random_state=0)\n\ntrain_ind, val_ind = next(splitter.split(df, groups=df[\"ancestor_id\"]))\n\ntrain_df = df.loc[train_ind].reset_index(drop=True)\nval_df = df.loc[val_ind].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:21:08.245907Z","iopub.execute_input":"2022-07-19T11:21:08.246784Z","iopub.status.idle":"2022-07-19T11:21:27.377186Z","shell.execute_reply.started":"2022-07-19T11:21:08.246750Z","shell.execute_reply":"2022-07-19T11:21:27.376158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_md = train_df[train_df[\"cell_type\"] == \"markdown\"].reset_index(drop=True)\nval_df_md = val_df[val_df[\"cell_type\"] == \"markdown\"].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:25:57.129529Z","iopub.execute_input":"2022-07-19T11:25:57.130588Z","iopub.status.idle":"2022-07-19T11:26:00.082435Z","shell.execute_reply.started":"2022-07-19T11:25:57.130555Z","shell.execute_reply":"2022-07-19T11:26:00.081481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_code = train_df[train_df[\"cell_type\"] == \"code\"].reset_index(drop=True)\nval_df_code = val_df[val_df[\"cell_type\"] == \"code\"].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:43:42.381167Z","iopub.execute_input":"2022-07-19T11:43:42.381750Z","iopub.status.idle":"2022-07-19T11:43:46.667071Z","shell.execute_reply.started":"2022-07-19T11:43:42.381717Z","shell.execute_reply":"2022-07-19T11:43:46.666071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_md","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:44:09.778297Z","iopub.execute_input":"2022-07-19T11:44:09.778670Z","iopub.status.idle":"2022-07-19T11:44:09.795825Z","shell.execute_reply.started":"2022-07-19T11:44:09.778634Z","shell.execute_reply":"2022-07-19T11:44:09.794670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## SAVE TRAIN-TEST MD-CODE DATASETS","metadata":{}},{"cell_type":"code","source":"train_df_md.to_csv('train_df_md.csv', index=False)\nval_df_md.to_csv('val_df_md.csv', index = False)\ntrain_df_code.to_csv('train_df_code.csv', index=False)\nval_df_code.to_csv('val_df_code.csv', index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LOADING","metadata":{}},{"cell_type":"code","source":"train_df_md = pd.read_csv('train_df_md.csv')\nval_df_md = pd.read_csv('val_df_md.csv')\nval_df_md = pd.read_csv('train_df_code.csv')\nval_df_code = pd.read_csv('val_df_code.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:05:30.353658Z","iopub.execute_input":"2022-07-19T12:05:30.354665Z","iopub.status.idle":"2022-07-19T12:06:07.480714Z","shell.execute_reply.started":"2022-07-19T12:05:30.354590Z","shell.execute_reply":"2022-07-19T12:06:07.479599Z"},"trusted":true},"execution_count":null,"outputs":[]}]}