{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport os\nimport codecs\nfrom typing import Dict, List, Tuple\nfrom text_unidecode import unidecode","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-26T00:43:50.985848Z","iopub.execute_input":"2022-07-26T00:43:50.986266Z","iopub.status.idle":"2022-07-26T00:43:51.002745Z","shell.execute_reply.started":"2022-07-26T00:43:50.986233Z","shell.execute_reply":"2022-07-26T00:43:51.001548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_path_2021 = '../input/feedback-prize-2021/'\ninput_path_2022 = '../input/feedback-prize-effectiveness/'","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:43:51.209752Z","iopub.execute_input":"2022-07-26T00:43:51.210394Z","iopub.status.idle":"2022-07-26T00:43:51.215258Z","shell.execute_reply.started":"2022-07-26T00:43:51.210362Z","shell.execute_reply":"2022-07-26T00:43:51.214144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_2021=pd.read_csv(input_path_2021 + 'train.csv')\ntrain_2022=pd.read_csv(input_path_2022 + 'train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:43:51.372722Z","iopub.execute_input":"2022-07-26T00:43:51.373462Z","iopub.status.idle":"2022-07-26T00:43:53.675824Z","shell.execute_reply.started":"2022-07-26T00:43:51.373428Z","shell.execute_reply":"2022-07-26T00:43:53.674666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/brandonhu0215/feedback-deberta-large-lb0-619\n\ndef replace_encoding_with_utf8(error: UnicodeError) -> Tuple[bytes, int]:\n    return error.object[error.start : error.end].encode(\"utf-8\"), error.end\n\n\ndef replace_decoding_with_cp1252(error: UnicodeError) -> Tuple[str, int]:\n    return error.object[error.start : error.end].decode(\"cp1252\"), error.end\n\ncodecs.register_error(\"replace_encoding_with_utf8\", replace_encoding_with_utf8)\ncodecs.register_error(\"replace_decoding_with_cp1252\", replace_decoding_with_cp1252)\n\n\ndef resolve_encodings_and_normalize(text: str) -> str:\n    text = (\n        text.encode(\"raw_unicode_escape\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n        .encode(\"cp1252\", errors=\"replace_encoding_with_utf8\")\n        .decode(\"utf-8\", errors=\"replace_decoding_with_cp1252\")\n    )\n    \n    text = unidecode(text)\n    \n    return text\n\n\ndef fetch_essay(essay_id: str, txt_dir: str):\n    essay_path = os.path.join(input_path_2021 + txt_dir, essay_id + '.txt')\n    # print(input_path_2022 + txt_dir, essay_id + '.txt')\n    essay_text = open(essay_path, 'r').read()\n    \n    return essay_text","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:44:41.735123Z","iopub.execute_input":"2022-07-26T00:44:41.735575Z","iopub.status.idle":"2022-07-26T00:44:41.747629Z","shell.execute_reply.started":"2022-07-26T00:44:41.735542Z","shell.execute_reply":"2022-07-26T00:44:41.746465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df = test_origin.copy()\n# SEP = CFG.tokenizer.sep_token\n\ntrain_2022['discourse_text'] = train_2022['discourse_text'].apply(resolve_encodings_and_normalize)\ntrain_2022['essay_text'] = train_2022['essay_id'].transform(fetch_essay, txt_dir='train')\ntrain_2022['essay_text'] = train_2022['essay_text'].apply(resolve_encodings_and_normalize)\ntrain_2022['text'] = train_2022['discourse_type'] + ' ' + train_2022['discourse_text'] + '[SEP]' + train_2022['essay_text']","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:47:08.204261Z","iopub.execute_input":"2022-07-26T00:47:08.204711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_2021['discourse_text'] = train_2021['discourse_text'].apply(resolve_encodings_and_normalize)\ntrain_2021['essay_text'] = train_2021['id'].transform(fetch_essay, txt_dir='train')\ntrain_2021['essay_text'] = train_2021['essay_text'].apply(resolve_encodings_and_normalize)\ntrain_2021['text'] = train_2021['discourse_type'] + ' ' + train_2021['discourse_text'] + '[SEP]' + train_2021['essay_text']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = train_2022[['text']]\ndf2 = train_2021[['text']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_df  = pd.concat([df1, df2]).drop_duplicates()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_df.to_csv('text.txt', header=False, index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}