{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport re\nimport pandas as pd\nfrom sklearn import model_selection\nfrom tqdm.auto import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-22T12:34:23.431457Z","iopub.execute_input":"2022-07-22T12:34:23.432571Z","iopub.status.idle":"2022-07-22T12:34:24.791943Z","shell.execute_reply.started":"2022-07-22T12:34:23.432458Z","shell.execute_reply":"2022-07-22T12:34:24.790700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:24.794330Z","iopub.execute_input":"2022-07-22T12:34:24.794935Z","iopub.status.idle":"2022-07-22T12:34:25.128874Z","shell.execute_reply.started":"2022-07-22T12:34:24.794886Z","shell.execute_reply":"2022-07-22T12:34:25.127458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_data(discourse_df, datatype=\"train\"):\n    idx = discourse_df.essay_id.values[0]\n    filename = os.path.join(\"../input/feedback-prize-effectiveness/\", datatype, idx + \".txt\")\n    with open(filename, \"r\") as f:\n        text = f.read()\n    min_idx = 0\n    starts = []\n    ends = []\n    for _, row in discourse_df.iterrows():\n        discourse_text = row[\"discourse_text\"]\n        matches = list(re.finditer(re.escape(discourse_text.strip()), text))\n        if len(matches) == 1:\n            discourse_start = matches[0].span()[0]\n            discourse_end = matches[0].span()[1]\n            min_idx = discourse_end\n        elif len(matches) > 1:\n            for match in matches:\n                discourse_start = match.span()[0]\n                discourse_end = match.span()[1]\n                if discourse_start >= min_idx:\n                    min_idx = discourse_end\n                    break\n        else:\n            discourse_start = -1\n            discourse_end = -1\n        starts.append(discourse_start)\n        ends.append(discourse_end)\n    discourse_df.loc[:, \"discourse_start\"] = starts\n    discourse_df.loc[:, \"discourse_end\"] = ends\n    return discourse_df","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:25.130278Z","iopub.execute_input":"2022-07-22T12:34:25.130594Z","iopub.status.idle":"2022-07-22T12:34:25.142545Z","shell.execute_reply.started":"2022-07-22T12:34:25.130565Z","shell.execute_reply":"2022-07-22T12:34:25.141364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_folds(data, num_splits):\n    data[\"kfold\"] = -1\n    \n    kf = model_selection.StratifiedGroupKFold(\n        n_splits=num_splits, shuffle=True, random_state=42\n    )\n\n    for f, (t_, v_) in enumerate(kf.split(X=data, y=data['discourse_effectiveness'].values, groups=data['essay_id'])):\n        data.loc[v_, 'kfold'] = f\n    \n    return data","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:25.145101Z","iopub.execute_input":"2022-07-22T12:34:25.146259Z","iopub.status.idle":"2022-07-22T12:34:25.155235Z","shell.execute_reply.started":"2022-07-22T12:34:25.146208Z","shell.execute_reply":"2022-07-22T12:34:25.154035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = []\nfor essay_id in tqdm(df.essay_id.unique(), total=len(df.essay_id.unique())):\n    temp_df = df[df.essay_id == essay_id].reset_index(drop=True)\n    res = create_data(temp_df, datatype=\"train\")\n    data.append(res)\n\ndata = pd.concat(data).reset_index(drop=True)\n\n# note:\n# only one training sample is missing\n# no samples are missing in the private test dataset\nprint(len(data[data.discourse_start == -1]))\ndata = data[data.discourse_start != -1].reset_index(drop=True)\ndata = create_folds(data, num_splits=5)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:25.156933Z","iopub.execute_input":"2022-07-22T12:34:25.157578Z","iopub.status.idle":"2022-07-22T12:36:01.779945Z","shell.execute_reply.started":"2022-07-22T12:34:25.157545Z","shell.execute_reply":"2022-07-22T12:36:01.778930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.to_csv(\"train_folds.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:36:01.781554Z","iopub.execute_input":"2022-07-22T12:36:01.782771Z","iopub.status.idle":"2022-07-22T12:36:02.344238Z","shell.execute_reply.started":"2022-07-22T12:36:01.782721Z","shell.execute_reply":"2022-07-22T12:36:02.342836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is based on work by @nbroad. Please give his notebook some love: https://www.kaggle.com/code/nbroad/token-classification-approach-fpe/","metadata":{}}]}