{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\n\nSEED = 22\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T17:06:28.331470Z","iopub.execute_input":"2022-07-13T17:06:28.332584Z","iopub.status.idle":"2022-07-13T17:06:28.372822Z","shell.execute_reply.started":"2022-07-13T17:06:28.332422Z","shell.execute_reply":"2022-07-13T17:06:28.371557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading data","metadata":{}},{"cell_type":"markdown","source":"CSV file containing correct ordering of the cells (markdown + code)","metadata":{}},{"cell_type":"code","source":"order_series = pd.read_csv(\"../input/AI4Code/train_orders.csv\", index_col='id', squeeze=True).str.split()\nprint(f\"Dataframe size: {len(order_series)}\")\norder_series.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:06:30.053498Z","iopub.execute_input":"2022-07-13T17:06:30.053896Z","iopub.status.idle":"2022-07-13T17:06:32.942812Z","shell.execute_reply.started":"2022-07-13T17:06:30.053863Z","shell.execute_reply":"2022-07-13T17:06:32.941576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Training json files containing markdown and code cells in a single dataframe","metadata":{}},{"cell_type":"code","source":"%%time\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed\n\ndef read_notebook(filename):\n    file_path = Path(os.path.join(train_data_path, filename))\n    df = pd.read_json(\n            file_path,\n            dtype={'cell_type': 'category', 'source': 'str'}) \\\n        .assign(notebook_id=file_path.stem) \\\n        .rename_axis('cell_id')\n    return df\n\ntrain_data_path = \"../input/AI4Code/train\"\n\njob = Parallel(n_jobs=-1)\nloaded_dfs = job(delayed(read_notebook)(filename) for filename in tqdm(list(os.listdir(train_data_path))[:]))\ncell_df = pd.concat(loaded_dfs, axis=0)\ndel loaded_dfs\n\nprint(f\"Dataset size: {len(cell_df)}\")\ncell_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:06:34.454884Z","iopub.execute_input":"2022-07-13T17:06:34.455670Z","iopub.status.idle":"2022-07-13T17:15:17.782309Z","shell.execute_reply.started":"2022-07-13T17:06:34.455628Z","shell.execute_reply":"2022-07-13T17:15:17.780892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Add order of cells in notebook","metadata":{}},{"cell_type":"code","source":"cell_df.insert(0, 'order', [order_series.loc[notebook_id].index(ind) for ind, notebook_id in tqdm(zip(list(cell_df.index), cell_df[\"notebook_id\"]), total=len(cell_df))])\ncell_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:15:17.785612Z","iopub.execute_input":"2022-07-13T17:15:17.787106Z","iopub.status.idle":"2022-07-13T17:17:20.620039Z","shell.execute_reply.started":"2022-07-13T17:15:17.787031Z","shell.execute_reply":"2022-07-13T17:17:20.618645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check memory usage","metadata":{}},{"cell_type":"code","source":"cell_df.info(memory_usage='deep')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T14:29:44.733483Z","iopub.execute_input":"2022-07-07T14:29:44.734944Z","iopub.status.idle":"2022-07-07T14:29:50.675630Z","shell.execute_reply.started":"2022-07-07T14:29:44.734856Z","shell.execute_reply":"2022-07-07T14:29:50.673467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Analysis for creating stratified column","metadata":{}},{"cell_type":"markdown","source":"1. Create cols for #markdown and #code cells","metadata":{}},{"cell_type":"code","source":"cell_count_df = pd.crosstab(cell_df[\"notebook_id\"], cell_df[\"cell_type\"]).reset_index()\ncell_count_df.columns = [\"notebook_id\", \"n_code\", \"n_markdown\"]\ncell_count_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:40:57.187060Z","iopub.execute_input":"2022-07-13T17:40:57.187996Z","iopub.status.idle":"2022-07-13T17:41:02.638989Z","shell.execute_reply.started":"2022-07-13T17:40:57.187947Z","shell.execute_reply":"2022-07-13T17:41:02.637749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2. Average number of characters in markdown and code cells","metadata":{}},{"cell_type":"code","source":"cell_df[\"n_char\"] = cell_df[\"source\"].map(lambda x: len(x.replace('\\n', '')))\nchar_series = cell_df.groupby([\"notebook_id\", \"cell_type\"])[\"n_char\"].mean()\nchar_df = char_series.unstack(level=1).reset_index()\nchar_df.columns = [\"notebook_id\", \"avg_code_len\", \"avg_markdown_len\"]\nchar_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:41:05.539938Z","iopub.execute_input":"2022-07-13T17:41:05.540362Z","iopub.status.idle":"2022-07-13T17:41:18.846061Z","shell.execute_reply.started":"2022-07-13T17:41:05.540327Z","shell.execute_reply":"2022-07-13T17:41:18.844563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_count_df = cell_count_df.merge(\n    char_df, \n    on=\"notebook_id\"\n)\ncell_count_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:41:22.864221Z","iopub.execute_input":"2022-07-13T17:41:22.864632Z","iopub.status.idle":"2022-07-13T17:41:23.147569Z","shell.execute_reply.started":"2022-07-13T17:41:22.864600Z","shell.execute_reply":"2022-07-13T17:41:23.146140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating bins","metadata":{}},{"cell_type":"markdown","source":"Code cell count","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(18, 5))\nplt.hist(cell_count_df[\"n_code\"], bins=30)\nplt.title(\"Code cell count histogram\")\nplt.xlabel(\"#code cells\")\nplt.ylabel(\"count\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:41:28.017067Z","iopub.execute_input":"2022-07-13T17:41:28.017526Z","iopub.status.idle":"2022-07-13T17:41:28.319531Z","shell.execute_reply.started":"2022-07-13T17:41:28.017492Z","shell.execute_reply":"2022-07-13T17:41:28.317571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins = [0, 25, 50, 100, 900]#[0, 20, 40, 60, 300]\ncell_count_df[\"binned_n_code\"] = pd.cut(cell_count_df[\"n_code\"], bins=bins, labels=bins[1:]).tolist()\ncell_count_df[\"binned_n_code\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:41:29.283831Z","iopub.execute_input":"2022-07-13T17:41:29.285000Z","iopub.status.idle":"2022-07-13T17:41:29.360992Z","shell.execute_reply.started":"2022-07-13T17:41:29.284953Z","shell.execute_reply":"2022-07-13T17:41:29.359679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Markdown cell count","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(18, 5))\nplt.hist(cell_count_df[\"n_markdown\"], bins=30)\nplt.title(\"Markdown cell count histogram\")\nplt.xlabel(\"#markdown cells\")\nplt.ylabel(\"count\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:41:31.284781Z","iopub.execute_input":"2022-07-13T17:41:31.285201Z","iopub.status.idle":"2022-07-13T17:41:31.554339Z","shell.execute_reply.started":"2022-07-13T17:41:31.285169Z","shell.execute_reply":"2022-07-13T17:41:31.552940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins = [0, 10, 30, 70, 600]#[0, 5, 15, 30, 50, 180]\ncell_count_df[\"binned_n_markdown\"] = pd.cut(cell_count_df[\"n_markdown\"], bins=bins, labels=bins[1:]).tolist()\ncell_count_df[\"binned_n_markdown\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:41:31.877980Z","iopub.execute_input":"2022-07-13T17:41:31.878404Z","iopub.status.idle":"2022-07-13T17:41:31.959832Z","shell.execute_reply.started":"2022-07-13T17:41:31.878372Z","shell.execute_reply":"2022-07-13T17:41:31.958478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Code cell avg. char length (The difference b/w min and max value is very high. So instead of plotting, calling describe)","metadata":{}},{"cell_type":"code","source":"cell_count_df[\"avg_code_len\"].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:41:32.476236Z","iopub.execute_input":"2022-07-13T17:41:32.476626Z","iopub.status.idle":"2022-07-13T17:41:32.497474Z","shell.execute_reply.started":"2022-07-13T17:41:32.476597Z","shell.execute_reply":"2022-07-13T17:41:32.496170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins = [0, 100, 150, 200, 300, 500, 350000]\ncell_count_df[\"binned_avg_code_len\"] = pd.cut(cell_count_df[\"avg_code_len\"], bins=bins, labels=bins[1:]).tolist()\ncell_count_df[\"binned_avg_code_len\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:41:32.598926Z","iopub.execute_input":"2022-07-13T17:41:32.599321Z","iopub.status.idle":"2022-07-13T17:41:32.681393Z","shell.execute_reply.started":"2022-07-13T17:41:32.599286Z","shell.execute_reply":"2022-07-13T17:41:32.680166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Markdown cell avg. char length (Adding this col to stratified col causes unique values to be created in the stratified col which do not allow train test split to work in a stratified fashion. So this col would not be included but it is here for analysis)","metadata":{}},{"cell_type":"code","source":"cell_count_df[\"avg_markdown_len\"].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:41:33.695784Z","iopub.execute_input":"2022-07-13T17:41:33.696191Z","iopub.status.idle":"2022-07-13T17:41:33.715116Z","shell.execute_reply.started":"2022-07-13T17:41:33.696159Z","shell.execute_reply":"2022-07-13T17:41:33.713465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins = [0, 40, 80, 150, 250, 400, 360000]\ncell_count_df[\"binned_avg_markdown_len\"] = pd.cut(cell_count_df[\"avg_markdown_len\"], bins=bins, labels=bins[1:]).tolist()\ncell_count_df[\"binned_avg_markdown_len\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:48:21.991614Z","iopub.execute_input":"2022-07-13T17:48:21.992076Z","iopub.status.idle":"2022-07-13T17:48:22.075242Z","shell.execute_reply.started":"2022-07-13T17:48:21.992040Z","shell.execute_reply":"2022-07-13T17:48:22.073964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_count_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:48:25.551173Z","iopub.execute_input":"2022-07-13T17:48:25.551950Z","iopub.status.idle":"2022-07-13T17:48:25.568332Z","shell.execute_reply.started":"2022-07-13T17:48:25.551910Z","shell.execute_reply":"2022-07-13T17:48:25.566635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create stratified column to split data","metadata":{}},{"cell_type":"code","source":"cell_count_df[\"stratified_markdown_code_cell\"] = cell_count_df[\"binned_n_code\"].astype(str) + \"-\" + \\\n                                                 cell_count_df[\"binned_n_markdown\"].astype(str) + \"-\" + \\\n                                                 cell_count_df[\"binned_avg_markdown_len\"].astype(str)\ncell_count_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:48:28.229301Z","iopub.execute_input":"2022-07-13T17:48:28.231167Z","iopub.status.idle":"2022-07-13T17:48:28.920122Z","shell.execute_reply.started":"2022-07-13T17:48:28.231037Z","shell.execute_reply":"2022-07-13T17:48:28.918365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check value count >= 2 for `stratified_markdown_code_cell` column so stratified splitting won't throw error","metadata":{}},{"cell_type":"code","source":"cell_count_df[\"stratified_markdown_code_cell\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:48:30.600259Z","iopub.execute_input":"2022-07-13T17:48:30.600683Z","iopub.status.idle":"2022-07-13T17:48:30.643822Z","shell.execute_reply.started":"2022-07-13T17:48:30.600648Z","shell.execute_reply":"2022-07-13T17:48:30.642880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Split the dataset","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_cell_count_df, val_cell_count_df = train_test_split(\n    cell_count_df, \n    test_size=0.03, \n    shuffle=True, stratify=cell_count_df[\"stratified_markdown_code_cell\"].values, \n    random_state=SEED\n)\n\nprint(f\"Train size: {len(train_cell_count_df)}\")\nprint(f\"Validation size: {len(val_cell_count_df)}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:48:37.926922Z","iopub.execute_input":"2022-07-13T17:48:37.927314Z","iopub.status.idle":"2022-07-13T17:48:39.318570Z","shell.execute_reply.started":"2022-07-13T17:48:37.927283Z","shell.execute_reply":"2022-07-13T17:48:39.317194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluation metric","metadata":{}},{"cell_type":"code","source":"from bisect import bisect\n\ndef count_inversions(a):\n    inversions = 0\n    sorted_so_far = []\n    for i, u in enumerate(a):  # O(N)\n        j = bisect(sorted_so_far, u)  # O(log N)\n        inversions += i - j\n        sorted_so_far.insert(j, u)  # O(N)\n    return inversions\n\ndef kendall_tau(ground_truth, predictions):\n    total_inversions = 0  # total inversions in predicted ranks across all instances\n    total_2max = 0  # maximum possible inversions across all instances\n    for gt, pred in zip(ground_truth, predictions):\n        ranks = [gt.index(x) for x in pred]  # rank predicted order in terms of ground truth\n        total_inversions += count_inversions(ranks)\n        n = len(gt)\n        total_2max += n * (n - 1)\n    return 1 - 4 * total_inversions / total_2max","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:48:48.105767Z","iopub.execute_input":"2022-07-13T17:48:48.106242Z","iopub.status.idle":"2022-07-13T17:48:48.116546Z","shell.execute_reply.started":"2022-07-13T17:48:48.106205Z","shell.execute_reply":"2022-07-13T17:48:48.115274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Sample submission df\n\nThe submission file should follow this format:","metadata":{}},{"cell_type":"code","source":"sample_sub_df = pd.read_csv(\"../input/AI4Code/sample_submission.csv\")\nsample_sub_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:48:51.639685Z","iopub.execute_input":"2022-07-13T17:48:51.640842Z","iopub.status.idle":"2022-07-13T17:48:51.661068Z","shell.execute_reply.started":"2022-07-13T17:48:51.640793Z","shell.execute_reply":"2022-07-13T17:48:51.659826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Trivial solution\n\nWithout applying any technique and keeping the order of markdown and code cells _as is_, perform evalution on the train and validation set ","metadata":{}},{"cell_type":"markdown","source":"Train","metadata":{}},{"cell_type":"code","source":"train_order_series = cell_df \\\n                    .loc[cell_df[\"notebook_id\"].isin(list(train_cell_count_df[\"notebook_id\"].unique()))] \\\n                    .groupby(\"notebook_id\") \\\n                    .apply(lambda x: list(x.index))\ntrain_order_series.index.names = [\"id\"]\ntrain_order_series.name = \"cell_order\"\ntrain_order_series.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:48:55.361448Z","iopub.execute_input":"2022-07-13T17:48:55.361843Z","iopub.status.idle":"2022-07-13T17:49:03.942012Z","shell.execute_reply.started":"2022-07-13T17:48:55.361812Z","shell.execute_reply":"2022-07-13T17:49:03.940647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_order_df = pd.merge(\n    train_order_series, \n    order_series[order_series.index.isin(list(train_cell_count_df[\"notebook_id\"].unique()))], \n    on=\"id\",\n    suffixes=(\"_prediction\", \"_true\")\n).head()\nassert train_order_df.isna().sum().sum() == 0\ntrain_order_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:49:33.161175Z","iopub.execute_input":"2022-07-13T17:49:33.161580Z","iopub.status.idle":"2022-07-13T17:49:34.110933Z","shell.execute_reply.started":"2022-07-13T17:49:33.161547Z","shell.execute_reply":"2022-07-13T17:49:34.108663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Validation","metadata":{}},{"cell_type":"code","source":"val_order_series = cell_df \\\n                    .loc[cell_df[\"notebook_id\"].isin(list(val_cell_count_df[\"notebook_id\"].unique()))] \\\n                    .groupby(\"notebook_id\") \\\n                    .apply(lambda x: list(x.index))\nval_order_series.index.names = [\"id\"]\nval_order_series.name = \"cell_order\"\nval_order_series.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:49:37.944554Z","iopub.execute_input":"2022-07-13T17:49:37.945628Z","iopub.status.idle":"2022-07-13T17:49:38.748274Z","shell.execute_reply.started":"2022-07-13T17:49:37.945577Z","shell.execute_reply":"2022-07-13T17:49:38.747060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_order_df = pd.merge(\n    val_order_series, \n    order_series[order_series.index.isin(list(val_cell_count_df[\"notebook_id\"].unique()))], \n    on=\"id\",\n    suffixes=(\"_prediction\", \"_true\")\n).head()\nassert val_order_df.isna().sum().sum() == 0\nval_order_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:49:46.024135Z","iopub.execute_input":"2022-07-13T17:49:46.024532Z","iopub.status.idle":"2022-07-13T17:49:46.082475Z","shell.execute_reply.started":"2022-07-13T17:49:46.024501Z","shell.execute_reply":"2022-07-13T17:49:46.081141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Evaluate","metadata":{}},{"cell_type":"code","source":"train_score = kendall_tau(\n    ground_truth=train_order_df[\"cell_order_true\"].tolist(), \n    predictions=train_order_df[\"cell_order_prediction\"].tolist()\n)\nval_score = kendall_tau(\n    ground_truth=val_order_df[\"cell_order_true\"].tolist(), \n    predictions=val_order_df[\"cell_order_prediction\"].tolist()\n)\n\nprint(f\"Train score: {train_score}\")\nprint(f\"Validation score: {val_score}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T17:49:50.503789Z","iopub.execute_input":"2022-07-13T17:49:50.504215Z","iopub.status.idle":"2022-07-13T17:49:50.515986Z","shell.execute_reply.started":"2022-07-13T17:49:50.504180Z","shell.execute_reply":"2022-07-13T17:49:50.514726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reference: \n\n- https://www.kaggle.com/code/ryanholbrook/getting-started-with-ai4code\n- https://www.kaggle.com/code/ryanholbrook/competition-metric-kendall-tau-correlation/notebook","metadata":{}}]}