{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport re\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport tensorflow_hub as hub\nfrom transformers import TFBertForSequenceClassification, BertTokenizer, TFBertModel\nfrom datasets import load_dataset, load_from_disk\nprint(\"tensorflow version:\",tf.__version__)\nprint(\"Num GPUs Available: \", len(tf.config.list_physical_devices('GPU')))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-23T14:51:41.007085Z","iopub.execute_input":"2022-08-23T14:51:41.007999Z","iopub.status.idle":"2022-08-23T14:51:55.074132Z","shell.execute_reply.started":"2022-08-23T14:51:41.007891Z","shell.execute_reply":"2022-08-23T14:51:55.072995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_essay(path, txt_folder, split_train):\n    df = pd.read_csv(path)\n    essay = []\n    essay_id = df[\"essay_id\"].to_numpy()\n    for i in range(len(essay_id)):\n        txt_path = os.path.join(txt_folder, essay_id[i]+\".txt\")\n        with open(txt_path,\"r\") as f:\n            txt = f.read()\n        txt = txt.replace(\"\\n\\n\",\" \\n\\n \")\n        txt = txt.encode(\"raw_unicode_escape\").decode(\"utf-8\", errors=\"ignore\")\n        essay.append(txt)\n    \n    df[\"essay\"] = essay\n    df[\"discourse_text\"] = df[\"discourse_type\"]+\" [SEP] \"+df[\"discourse_text\"]\n    \n    if split_train == True:\n        essay_id_lst = np.unique(essay_id)\n        np.random.shuffle(essay_id_lst)\n        train = df[df[\"essay_id\"].isin(essay_id_lst[:int(len(essay_id_lst)*0.8)])]\n        test = df[df[\"essay_id\"].isin(essay_id_lst[int(len(essay_id_lst)*0.8):])]\n        df.to_csv(\"df.csv\", index=False)\n        train.to_csv(\"train.csv\", index=False)\n        test.to_csv(\"test.csv\", index=False)\n        \n    else: df.to_csv(\"df.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:52:04.168126Z","iopub.execute_input":"2022-08-23T14:52:04.168829Z","iopub.status.idle":"2022-08-23T14:52:04.178930Z","shell.execute_reply.started":"2022-08-23T14:52:04.168778Z","shell.execute_reply":"2022-08-23T14:52:04.177866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"add_essay(path=\"../input/feedback-prize-effectiveness/train.csv\", \n          txt_folder=\"../input/feedback-prize-effectiveness/train\", split_train=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:52:05.731983Z","iopub.execute_input":"2022-08-23T14:52:05.732985Z","iopub.status.idle":"2022-08-23T14:52:36.922572Z","shell.execute_reply.started":"2022-08-23T14:52:05.732945Z","shell.execute_reply":"2022-08-23T14:52:36.921589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds=tf.data.experimental.CsvDataset(\"df.csv\", record_defaults=[tf.string]*6, header=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:52:36.924695Z","iopub.execute_input":"2022-08-23T14:52:36.925125Z","iopub.status.idle":"2022-08-23T14:52:36.935156Z","shell.execute_reply.started":"2022-08-23T14:52:36.925087Z","shell.execute_reply":"2022-08-23T14:52:36.934137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in ds.take(1):\n    print(i[0].numpy())\n    print(i[1].numpy())\n    print(i[2].numpy())\n    print(i[3].numpy())\n    print(i[4].numpy())\n    print(i[5].numpy())","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:52:36.936919Z","iopub.execute_input":"2022-08-23T14:52:36.937272Z","iopub.status.idle":"2022-08-23T14:52:36.974218Z","shell.execute_reply.started":"2022-08-23T14:52:36.937238Z","shell.execute_reply":"2022-08-23T14:52:36.972714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def statistic():\n    ID=[]\n    txt=[]\n    essay=[]\n    length=[]\n    topic=[]\n    score=[]\n\n    for i in ds:\n        a=i[0].numpy()\n        b=i[1].numpy()\n        x=len(i[2].numpy().split())\n        y=i[3].numpy().decode('utf-8')\n        z=i[4].numpy().decode('utf-8')\n        c=len(i[5].numpy().split())\n\n        ID.append(a)\n        txt.append(b)\n        essay.append(c)\n        length.append(x)\n        topic.append(y)\n        if z=='Adequate':\n            score.append(0)\n        elif z=='Ineffective':\n            score.append(-1)\n        else: score.append(1)\n\n    statistic_frame=pd.DataFrame({\n        \"id\":ID,\n        \"txt\":txt,\n        \"essay\":essay,\n        \"length\":length,\n        \"topic\":topic,\n        \"score\":score\n    })\n    \n    return statistic_frame\n\nstatistic_frame=statistic()\nstatistic_frame[['length','essay','score']].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:52:36.976696Z","iopub.execute_input":"2022-08-23T14:52:36.977220Z","iopub.status.idle":"2022-08-23T14:52:43.968584Z","shell.execute_reply.started":"2022-08-23T14:52:36.977184Z","shell.execute_reply":"2022-08-23T14:52:43.967688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_count = statistic_frame.groupby(\"score\").count()[\"id\"].to_numpy()\nscore_count = (score_count/score_count.sum())*100\nscore_label = [f'Ineffective: {score_count[0]:.2f}%',\n               f'Adequate: {score_count[1]:.2f}%',\n               f'Effective: {score_count[2]:.2f}%']\n\nplt.pie(score_count, labels = score_label)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:52:43.970822Z","iopub.execute_input":"2022-08-23T14:52:43.971418Z","iopub.status.idle":"2022-08-23T14:52:44.103526Z","shell.execute_reply.started":"2022-08-23T14:52:43.971381Z","shell.execute_reply":"2022-08-23T14:52:44.101977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(16,6))\n\ntxt_q99=np.quantile(statistic_frame[\"length\"], 0.99)\ntxt_q100=max(statistic_frame[\"length\"])\n\nessay_q99=np.quantile(statistic_frame[\"essay\"], 0.99)\nessay_q100=max(statistic_frame[\"essay\"])\n\nax[0].hist(statistic_frame[\"length\"],bins=100)\nax[0].plot([txt_q99,txt_q99],[0,1500],lw=1,color=\"k\",ls='--')\nax[0].text(txt_q99*0.8,1500*1.1,\"99th quantile: %s words\"%int(txt_q99))\nax[0].plot([txt_q100,txt_q100],[0,1000],lw=1,color=\"k\",ls='--')\nax[0].text(txt_q100*0.7,1000*1.1,\"the longest: %s words\"%int(txt_q100))\nax[0].title.set_text(\"disclosure words\")\n\nax[1].hist(statistic_frame[\"essay\"],bins=100,color=\"g\")\nax[1].plot([essay_q99,essay_q99],[0,1500],lw=1,color=\"k\",ls='--')\nax[1].text(essay_q99*0.8,1500*1.1,\"99th quantile: %s words\"%int(essay_q99))\nax[1].plot([essay_q100,essay_q100],[0,1000],lw=1,color=\"k\",ls='--')\nax[1].text(essay_q100*0.7,1000*1.1,\"the longest: %s words\"%int(essay_q100))\nax[1].title.set_text(\"essay words\")\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:52:44.105254Z","iopub.execute_input":"2022-08-23T14:52:44.105822Z","iopub.status.idle":"2022-08-23T14:52:44.997645Z","shell.execute_reply.started":"2022-08-23T14:52:44.105766Z","shell.execute_reply":"2022-08-23T14:52:44.996721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://www.delftstack.com/zh-tw/howto/python/moving-average-python/\ndef moving_average(x, w):\n    return np.convolve(x, np.ones(w), 'valid') / w\n\nl=statistic_frame.sort_values(\"length\")[\"length\"].to_numpy()\ns=statistic_frame.sort_values(\"length\")[\"score\"].to_numpy()\n\nma_l=moving_average(l, int(s.shape[0]/100))\nma_s=moving_average(s, int(s.shape[0]/100))\n\nmax_s=max(ma_s)\nmax_l=ma_l[np.where(ma_s==max_s)[0]][0]\n\nplt.ylim(-0.1, max_s*1.15)\nplt.plot(ma_l,ma_s)\nplt.plot(max_l,max_s,marker=\"o\",markersize=4,color=\"k\")\nplt.text(max_l*0.1,max_s*1.06,\"on average, a {0:.2f}-word essay has the highest score {1:.2f}\".format(max_l, max_s))\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:52:44.999386Z","iopub.execute_input":"2022-08-23T14:52:44.999768Z","iopub.status.idle":"2022-08-23T14:52:45.221248Z","shell.execute_reply.started":"2022-08-23T14:52:44.999732Z","shell.execute_reply":"2022-08-23T14:52:45.220170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"statistic=statistic_frame.groupby(by=\"topic\").agg({\"length\":[\"mean\",\"count\"],\"essay\":[\"mean\",\"count\"],\"score\":[\"mean\",\"std\"]})\nstatistic.sort_values((\"length\",\"count\"),ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:52:45.222966Z","iopub.execute_input":"2022-08-23T14:52:45.223588Z","iopub.status.idle":"2022-08-23T14:52:45.254851Z","shell.execute_reply.started":"2022-08-23T14:52:45.223548Z","shell.execute_reply":"2022-08-23T14:52:45.253957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_train = tf.data.experimental.CsvDataset(\"train.csv\", record_defaults=[tf.string]*6, header=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:52:45.256563Z","iopub.execute_input":"2022-08-23T14:52:45.256976Z","iopub.status.idle":"2022-08-23T14:52:45.263526Z","shell.execute_reply.started":"2022-08-23T14:52:45.256938Z","shell.execute_reply":"2022-08-23T14:52:45.262468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_lst = [\"Ineffective\",\"Adequate\",\"Effective\"]\n@tf.function\ndef one_hot_score(x, score_lst = score_lst):\n    output = [0]*len(score_lst)\n    for i in range(len(score_lst)):\n        if score_lst[i] == x:\n            output[i] = 1\n    return output","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:52:45.267715Z","iopub.execute_input":"2022-08-23T14:52:45.268686Z","iopub.status.idle":"2022-08-23T14:52:45.275455Z","shell.execute_reply.started":"2022-08-23T14:52:45.268641Z","shell.execute_reply":"2022-08-23T14:52:45.274364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = ds_train.map(lambda x0, x1, x2, x3, x4, x5: [x0])\nlabel = ds_train.map(lambda x0, x1, x2, x3, x4, x5: [one_hot_score(x4)])\n\nfor i in dataset.batch(2).take(1):\n    print(i)\nprint(\"\")\nfor i in label.batch(2).take(1):\n    print(i)\ndel i","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:52:45.276782Z","iopub.execute_input":"2022-08-23T14:52:45.277311Z","iopub.status.idle":"2022-08-23T14:52:45.535102Z","shell.execute_reply.started":"2022-08-23T14:52:45.277274Z","shell.execute_reply":"2022-08-23T14:52:45.534079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_token = load_dataset('csv',data_files=\"train.csv\",split=\"train\",keep_in_memory=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:52:45.536779Z","iopub.execute_input":"2022-08-23T14:52:45.537156Z","iopub.status.idle":"2022-08-23T14:53:07.005862Z","shell.execute_reply.started":"2022-08-23T14:52:45.537121Z","shell.execute_reply":"2022-08-23T14:53:07.004562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import DataCollatorWithPadding\nmax_len = 512\ntokenizer = BertTokenizer.from_pretrained(\"../input/bert-en-cased-l12-h768-a12-4/huggingface_bert_base_cased/huggingface_bert_base_cased\")\ndata_collator = DataCollatorWithPadding(tokenizer=tokenizer,max_length=max_len, padding=\"max_length\" , return_tensors=\"tf\")","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:53:07.011587Z","iopub.execute_input":"2022-08-23T14:53:07.014242Z","iopub.status.idle":"2022-08-23T14:53:07.090734Z","shell.execute_reply.started":"2022-08-23T14:53:07.014200Z","shell.execute_reply":"2022-08-23T14:53:07.089717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tokenizer.convert_ids_to_tokens(tokenizer(\"english\")['input_ids']))\nprint(tokenizer.convert_ids_to_tokens(tokenizer(\"English\")['input_ids']))","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:53:07.095341Z","iopub.execute_input":"2022-08-23T14:53:07.097500Z","iopub.status.idle":"2022-08-23T14:53:07.107451Z","shell.execute_reply.started":"2022-08-23T14:53:07.097464Z","shell.execute_reply":"2022-08-23T14:53:07.106308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_discourse = ds_token.map(lambda x: tokenizer(x[\"discourse_text\"], max_length=max_len, truncation=True, padding=\"max_length\"))\nds_essay = ds_token.map(lambda x: tokenizer(x[\"essay\"], max_length=max_len, truncation=True, padding=\"max_length\"))","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:53:07.112225Z","iopub.execute_input":"2022-08-23T14:53:07.112762Z","iopub.status.idle":"2022-08-23T14:59:11.672563Z","shell.execute_reply.started":"2022-08-23T14:53:07.112729Z","shell.execute_reply":"2022-08-23T14:59:11.671518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"discourse = ds_discourse.to_tf_dataset(\n    columns=['input_ids', 'token_type_ids', 'attention_mask'],\n    batch_size=1,\n    shuffle=False,\n    collate_fn=data_collator)\ndiscourse = discourse.map(lambda x: (x[\"input_ids\"],x[\"token_type_ids\"],x[\"attention_mask\"]))\n\nessay = ds_essay.to_tf_dataset(\n    columns=['input_ids', 'token_type_ids', 'attention_mask'],\n    batch_size=1,\n    shuffle=False,\n    collate_fn=data_collator)\nessay = essay.map(lambda x: (x[\"input_ids\"],x[\"token_type_ids\"],x[\"attention_mask\"]))","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:59:11.676155Z","iopub.execute_input":"2022-08-23T14:59:11.677044Z","iopub.status.idle":"2022-08-23T14:59:13.129825Z","shell.execute_reply.started":"2022-08-23T14:59:11.677004Z","shell.execute_reply":"2022-08-23T14:59:13.128860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = tf.data.Dataset.zip(((dataset, discourse, essay), label))\ntrain_dataset = train_dataset.map(lambda x, y: (\n    (x[0][0],\n     tf.squeeze(x[1][0]),tf.squeeze(x[1][1]),tf.squeeze(x[1][2]),\n     tf.squeeze(x[2][0]),tf.squeeze(x[2][2]),tf.squeeze(x[2][2])),y),num_parallel_calls=tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:59:13.131363Z","iopub.execute_input":"2022-08-23T14:59:13.132075Z","iopub.status.idle":"2022-08-23T14:59:13.196680Z","shell.execute_reply.started":"2022-08-23T14:59:13.132038Z","shell.execute_reply":"2022-08-23T14:59:13.195843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for j in ds_train.batch(2).take(1):\n    print(\"first essay:\\n\",j[2][0].numpy().decode('utf-8'))\n    print(\"\\n\",j[-1][0].numpy().decode('utf-8'))\n    print(\"score:\",j[-2][0].numpy().decode('utf-8'),\"\\n\")\n    print(\"second essay:\\n\",j[2][1].numpy().decode('utf-8'))\n    print(\"\\n\",j[-1][1].numpy().decode('utf-8'))\n    print(\"score:\",j[-2][1].numpy().decode('utf-8'),\"\\n\")\n    \ndel j","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:59:13.198146Z","iopub.execute_input":"2022-08-23T14:59:13.198574Z","iopub.status.idle":"2022-08-23T14:59:13.221037Z","shell.execute_reply.started":"2022-08-23T14:59:13.198535Z","shell.execute_reply":"2022-08-23T14:59:13.220068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in train_dataset.batch(2).take(1):\n    pass\nprint(\"first essay:\\n\",i[0][1][0][i[0][1][0].numpy()>0])\nprint(np.array(tokenizer.convert_ids_to_tokens(i[0][1][0][i[0][1][0].numpy()>0])))\nprint(np.array(tokenizer.convert_ids_to_tokens(i[0][4][0][i[0][4][0].numpy()>0])))\nprint(\"score:\",i[1][0][0].numpy(),\"\\n\")\n\nprint(\"second essay:\\n\",i[0][1][1][i[0][1][1].numpy()>0])\nprint(np.array(tokenizer.convert_ids_to_tokens(i[0][1][1][i[0][1][1].numpy()>0])))\nprint(np.array(tokenizer.convert_ids_to_tokens(i[0][4][1][i[0][4][1].numpy()>0])))\nprint(\"score:\",i[1][0][1].numpy(),\"\\n\")\n\ndel i","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:59:13.224136Z","iopub.execute_input":"2022-08-23T14:59:13.224828Z","iopub.status.idle":"2022-08-23T14:59:13.879575Z","shell.execute_reply.started":"2022-08-23T14:59:13.224768Z","shell.execute_reply":"2022-08-23T14:59:13.878454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_batch = 32\nnum_shuffle = int(discourse.cardinality().numpy()/num_batch)\ntrain_dataset = train_dataset.batch(num_batch).shuffle(num_shuffle)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:59:13.881270Z","iopub.execute_input":"2022-08-23T14:59:13.881918Z","iopub.status.idle":"2022-08-23T14:59:13.891012Z","shell.execute_reply.started":"2022-08-23T14:59:13.881874Z","shell.execute_reply":"2022-08-23T14:59:13.890030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_pipeline(df_path, set_label=True, max_len=512, num_batch=64):\n    \n    if set_label == True:\n        dataset = tf.data.experimental.CsvDataset(df_path, record_defaults=[tf.string]*6, header=True)\n        ds = dataset.map(lambda x0, x1, x2, x3, x4, x5: [x0])\n        label = dataset.map(lambda x0, x1, x2, x3, x4, x5: [one_hot_score(x4)])\n    else: \n        dataset = tf.data.experimental.CsvDataset(df_path, record_defaults=[tf.string]*5, header=True)\n        ds = dataset.map(lambda x0, x1, x2, x3, x4: [x0])\n    \n    ds_token = load_dataset('csv', data_files=df_path, split=\"train\", keep_in_memory=False)\n    tokenizer = BertTokenizer.from_pretrained(\"../input/bert-en-cased-l12-h768-a12-4/huggingface_bert_base_cased/huggingface_bert_base_cased\")\n    data_collator = DataCollatorWithPadding(tokenizer=tokenizer,max_length=max_len, padding=\"max_length\" , return_tensors=\"tf\")\n    \n    ds_discourse = ds_token.map(lambda x: tokenizer(x[\"discourse_text\"], max_length=max_len, truncation=True, padding=\"max_length\"))\n    ds_essay = ds_token.map(lambda x: tokenizer(x[\"essay\"], max_length=max_len, truncation=True, padding=\"max_length\"))\n    \n    discourse = ds_discourse.to_tf_dataset(\n        columns=['input_ids', 'token_type_ids', 'attention_mask'],\n        batch_size=1,\n        shuffle=False,\n        collate_fn=data_collator)\n    discourse = discourse.map(lambda x: (x[\"input_ids\"],x[\"token_type_ids\"],x[\"attention_mask\"]))\n\n    essay = ds_essay.to_tf_dataset(\n        columns=['input_ids', 'token_type_ids', 'attention_mask'],\n        batch_size=1,\n        shuffle=False,\n        collate_fn=data_collator)\n    essay = essay.map(lambda x: (x[\"input_ids\"],x[\"token_type_ids\"],x[\"attention_mask\"]))\n\n    if set_label == True:\n        output = tf.data.Dataset.zip(((ds,discourse,essay), label))\n        output = output.map(lambda x, y: ((\n           (x[0][0],tf.squeeze(x[1][0]),tf.squeeze(x[1][1]),tf.squeeze(x[1][2]),\n            tf.squeeze(x[2][0]),tf.squeeze(x[2][2]),tf.squeeze(x[2][2])),y)),num_parallel_calls=tf.data.AUTOTUNE)\n        output = output.batch(num_batch)\n    else:\n        output = tf.data.Dataset.zip((ds,discourse,essay))\n        output = output.map(lambda x, y, z: (x[0],tf.squeeze(y[0]),tf.squeeze(y[1]),tf.squeeze(y[2]),\n                            tf.squeeze(z[0]),tf.squeeze(z[1]),tf.squeeze(z[2])),\n            num_parallel_calls=tf.data.AUTOTUNE)\n        output = output.batch(num_batch)\n        \n    return output","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:59:13.896086Z","iopub.execute_input":"2022-08-23T14:59:13.896342Z","iopub.status.idle":"2022-08-23T14:59:13.912219Z","shell.execute_reply.started":"2022-08-23T14:59:13.896318Z","shell.execute_reply":"2022-08-23T14:59:13.911299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = data_pipeline(\"test.csv\", num_batch=32)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:59:13.915278Z","iopub.execute_input":"2022-08-23T14:59:13.915537Z","iopub.status.idle":"2022-08-23T15:01:00.452616Z","shell.execute_reply.started":"2022-08-23T14:59:13.915514Z","shell.execute_reply":"2022-08-23T15:01:00.451652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count=0\nfor _ in test_dataset:\n    count=count+1\ncount","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:01:00.454095Z","iopub.execute_input":"2022-08-23T15:01:00.454429Z","iopub.status.idle":"2022-08-23T15:01:14.965724Z","shell.execute_reply.started":"2022-08-23T15:01:00.454394Z","shell.execute_reply":"2022-08-23T15:01:14.964731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in test_dataset.take(1):\n    pass\ntest_df = pd.read_csv(\"test.csv\")\nk0=0\nk1=2\ntest_df[test_df[\"discourse_id\"].isin([j.decode('utf-8') for j in i[0][0][:k1+1].numpy()])]","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:01:14.967144Z","iopub.execute_input":"2022-08-23T15:01:14.968167Z","iopub.status.idle":"2022-08-23T15:01:15.258670Z","shell.execute_reply.started":"2022-08-23T15:01:14.968129Z","shell.execute_reply":"2022-08-23T15:01:15.257712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"discourse_text:\")\nprint(\"lst essay:\",np.array(tokenizer.convert_ids_to_tokens(i[0][1][k0].numpy()[i[0][1][k0].numpy()>0])),\"\\n\")\nprint(\"3rd essay:\",np.array(tokenizer.convert_ids_to_tokens(i[0][1][k1].numpy()[i[0][1][k1].numpy()>0])),\"\\n\")\nprint(\"score:\")\nprint(\"lst essay:\",i[1][0][k0].numpy())\nprint(\"3rd essay:\",i[1][0][k1].numpy())","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:01:15.260089Z","iopub.execute_input":"2022-08-23T15:01:15.260419Z","iopub.status.idle":"2022-08-23T15:01:15.272353Z","shell.execute_reply.started":"2022-08-23T15:01:15.260393Z","shell.execute_reply":"2022-08-23T15:01:15.271217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def bert_model():\n    bert_layer = hub.KerasLayer(\"../input/bert-en-cased-l12-h768-a12-4/bert_en_cased_L-12_H-768_A-12_4\",trainable=False,name='BERT')\n\n    ids = tf.keras.Input(shape=(1,), dtype=tf.string)\n\n    disclosure_input_ids = tf.keras.Input(shape=(512,), name=\"disclosure_input_ids\", dtype=tf.int32)\n    disclosure_token_type_ids = tf.keras.Input(shape=(512,), name=\"disclosure_token_type_ids\", dtype=tf.int32)\n    disclosure_attention_mask = tf.keras.Input(shape=(512,), name=\"disclosure_attention_mask\", dtype=tf.int32)\n    \n    essay_input_ids = tf.keras.Input(shape=(512,), name=\"essay_input_ids\", dtype=tf.int32)\n    essay_token_type_ids = tf.keras.Input(shape=(512,), name=\"essay_token_type_ids\", dtype=tf.int32)\n    essay_attention_mask = tf.keras.Input(shape=(512,), name=\"essay_attention_mask\", dtype=tf.int32)\n\n    disclosure = bert_layer({\"input_mask\": disclosure_attention_mask,\n                                  \"input_type_ids\": disclosure_token_type_ids,\n                                  \"input_word_ids\": disclosure_input_ids})['sequence_output']\n    \n    essay = bert_layer({\"input_mask\": essay_attention_mask,\n                              \"input_type_ids\": essay_token_type_ids,\n                              \"input_word_ids\": essay_input_ids})['sequence_output']\n    \n    disclosure = tf.keras.layers.Conv1D(filters=128, kernel_size=1)(disclosure)\n    disclosure = tf.keras.layers.Conv1D(filters=128, kernel_size=3, strides=2)(disclosure)\n    disclosure = tf.keras.layers.LeakyReLU(name=\"disclosure\")(disclosure)\n\n    essay = tf.keras.layers.Conv1D(filters=128, kernel_size=1)(essay)\n    essay = tf.keras.layers.Conv1D(filters=128, kernel_size=3, strides=2)(essay)\n    essay = tf.keras.layers.LeakyReLU(name=\"essay\")(essay)\n    \n    text = tf.keras.layers.concatenate([disclosure, essay], axis=-2)\n    \n    x = tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(128, return_sequences=False, dropout=0.2))(text)\n    \n    x = tf.keras.layers.Dense(int(x.shape[-1]/2), activation=\"relu\")(x)\n    x = tf.keras.layers.Dropout(0.2)(x)\n    x = tf.keras.layers.Dense(int(x.shape[-1]/2))(x)\n    x = tf.keras.layers.Dropout(0.1)(x)\n    \n    output = tf.keras.layers.Dense(3, activation=\"softmax\", name=\"output\")(x)\n    embedding_model = tf.keras.Model((ids, disclosure_input_ids, disclosure_token_type_ids, disclosure_attention_mask,\n                                     essay_input_ids, essay_token_type_ids, essay_attention_mask), \n                                     output)\n    return embedding_model","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:01:15.274069Z","iopub.execute_input":"2022-08-23T15:01:15.274608Z","iopub.status.idle":"2022-08-23T15:01:15.290426Z","shell.execute_reply.started":"2022-08-23T15:01:15.274575Z","shell.execute_reply":"2022-08-23T15:01:15.289525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=bert_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:01:15.291918Z","iopub.execute_input":"2022-08-23T15:01:15.292571Z","iopub.status.idle":"2022-08-23T15:01:29.717054Z","shell.execute_reply.started":"2022-08-23T15:01:15.292538Z","shell.execute_reply":"2022-08-23T15:01:29.716079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights(\"../input/feedback-prize-weights/model\")","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:01:29.718476Z","iopub.execute_input":"2022-08-23T15:01:29.720065Z","iopub.status.idle":"2022-08-23T15:01:32.744766Z","shell.execute_reply.started":"2022-08-23T15:01:29.720024Z","shell.execute_reply":"2022-08-23T15:01:32.743740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(model, show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:01:32.746360Z","iopub.execute_input":"2022-08-23T15:01:32.746730Z","iopub.status.idle":"2022-08-23T15:01:34.095920Z","shell.execute_reply.started":"2022-08-23T15:01:32.746695Z","shell.execute_reply":"2022-08-23T15:01:34.094830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"steps = num_shuffle\ntotal_epoch = 35\n\ndef lr_scheduler(epoch, lr, total_epoch=total_epoch):\n    output = lr*tf.math.exp(-(epoch/(total_epoch+1)))\n    if output < 2.5e-5:\n        output = 2.5e-5\n    return output\n\noptimizer = tf.keras.optimizers.Adam(learning_rate=1e-4)\n\nmodel.compile(optimizer = optimizer, \n              loss=\"categorical_crossentropy\", \n              metrics=[\"categorical_accuracy\"])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:05:02.708470Z","iopub.execute_input":"2022-08-23T15:05:02.709571Z","iopub.status.idle":"2022-08-23T15:05:02.729125Z","shell.execute_reply.started":"2022-08-23T15:05:02.709522Z","shell.execute_reply":"2022-08-23T15:05:02.728052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ouput = model.evaluate(test_dataset)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:05:04.238669Z","iopub.execute_input":"2022-08-23T15:05:04.239279Z","iopub.status.idle":"2022-08-23T15:10:28.611318Z","shell.execute_reply.started":"2022-08-23T15:05:04.239242Z","shell.execute_reply":"2022-08-23T15:10:28.610250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if True:\n    callback_checkpoint = tf.keras.callbacks.ModelCheckpoint(filepath=\"model\",save_weights_only=True,save_freq=\"epoch\")\n    callback_earlystopping = tf.keras.callbacks.EarlyStopping(patience=3, monitor='val_loss')\n    callback_scheduler = tf.keras.callbacks.LearningRateScheduler(lr_scheduler, verbose=1)\n\n    his = model.fit(train_dataset, epochs=total_epoch, validation_data=test_dataset, \n                    callbacks=[callback_checkpoint, callback_earlystopping, callback_scheduler])","metadata":{"execution":{"iopub.status.busy":"2022-08-23T09:08:08.648468Z","iopub.execute_input":"2022-08-23T09:08:08.648840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    read_his = his\n\n    fig, ax = plt.subplots(1, 3, figsize=(21,5))\n\n    ax[0].set_title(\"accuracy\")\n    ax[0].plot(read_his['categorical_accuracy'], c=\"b\", label=\"training\")\n    ax[0].plot(read_his['val_categorical_accuracy'], c=\"g\", label=\"validation\")\n    ax[0].legend(loc=\"lower right\")\n    ax[0].set_xticks(list(range(1,len(read_his[\"loss\"]),2)))\n    ax[0].set_xticklabels(list(range(1,len(read_his[\"loss\"]),2)))\n\n    ax[1].set_title(\"loss\")\n    ax[1].plot(read_his['loss'], c=\"b\", label=\"training\")\n    ax[1].plot(read_his['val_loss'], c=\"g\", label=\"validation\")\n    ax[1].legend(loc=\"upper right\")\n    ax[1].set_xticks(list(range(1,len(read_his[\"loss\"]),2)))\n    ax[1].set_xticklabels(list(range(1,len(read_his[\"loss\"]),2)))\n\n    ax[2].set_title(\"learning rate\")\n    ax[2].plot(read_his['lr'])\n    ax[2].set_xticks(list(range(1,len(read_his[\"loss\"]),2)))\n    ax[2].set_xticklabels(list(range(1,len(read_his[\"loss\"]),2)))\n\n    fig.show()\nexcept: pass","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:04:27.890121Z","iopub.execute_input":"2022-08-23T15:04:27.890771Z","iopub.status.idle":"2022-08-23T15:04:27.903920Z","shell.execute_reply.started":"2022-08-23T15:04:27.890732Z","shell.execute_reply":"2022-08-23T15:04:27.902870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model_performance(model=model,test_dataset=test_dataset):\n    output = []\n    label = []\n    \n    for data in test_dataset:\n        test_output = model(data[0])\n        for i in test_output:\n            output.append(i)\n        for i in data[1][0]:\n            label.append(i)\n            \n    output = np.array(output)\n    label = np.array(label)\n    \n    return output, label\n\noutput, label = model_performance()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:10:44.850882Z","iopub.execute_input":"2022-08-23T15:10:44.851482Z","iopub.status.idle":"2022-08-23T15:16:02.884183Z","shell.execute_reply.started":"2022-08-23T15:10:44.851445Z","shell.execute_reply":"2022-08-23T15:16:02.883142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def output_to_score(output=output):\n    score = []\n    for i in output:\n        highest = np.where(i == np.max(i))[0][0]\n        lst = [0]*3\n        lst[highest]=1\n        score.append(lst)\n        \n    return np.array(score)\n\noutput_score = output_to_score(output)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:16:39.851004Z","iopub.execute_input":"2022-08-23T15:16:39.851365Z","iopub.status.idle":"2022-08-23T15:16:39.914744Z","shell.execute_reply.started":"2022-08-23T15:16:39.851334Z","shell.execute_reply":"2022-08-23T15:16:39.913827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def label_inverse(label=label):\n    ineffective = []\n    adequeate = []\n    effective = []\n    count_label = 0\n    for i in label:\n        if (i == [1,0,0]).all(): ineffective.append(count_label)\n        elif (i == [0,1,0]).all(): adequeate.append(count_label)\n        elif (i == [0,0,1]).all(): effective.append(count_label)\n        count_label = count_label+1\n        \n    label_dict = dict({\"ineffective\":np.array(ineffective),\"adequeate\":np.array(adequeate),\"effective\":np.array(effective)})\n    \n    return label_dict\n\nlabel_dict = label_inverse()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:16:41.845324Z","iopub.execute_input":"2022-08-23T15:16:41.845892Z","iopub.status.idle":"2022-08-23T15:16:41.961053Z","shell.execute_reply.started":"2022-08-23T15:16:41.845846Z","shell.execute_reply":"2022-08-23T15:16:41.960064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in label_dict.keys():\n    print(i,\":\",label_dict[i].shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:16:43.718793Z","iopub.execute_input":"2022-08-23T15:16:43.719408Z","iopub.status.idle":"2022-08-23T15:16:43.728676Z","shell.execute_reply.started":"2022-08-23T15:16:43.719354Z","shell.execute_reply":"2022-08-23T15:16:43.727348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ineffective_ineffective = []\nineffective_adequeate = []\nineffective_effective = []\n\nadequeate_ineffective = []\nadequeate_adequeate = []\nadequeate_effective = []\n\neffective_ineffective = []\neffective_adequeate = []\neffective_effective = []\n\ncount = 0\nfor i in label[label_dict[\"ineffective\"]] - output_score[label_dict[\"ineffective\"]]:\n    if (i == [0, 0, 0]).all(): ineffective_ineffective.append(count)\n    elif (i == [1, -1, 0]).all(): ineffective_adequeate.append(count)\n    elif (i == [1, 0, -1]).all(): ineffective_adequeate.append(count)\n    count = count+1\n    \ncount = 0\nfor i in label[label_dict[\"adequeate\"]] - output_score[label_dict[\"adequeate\"]]:\n    if (i == [0, 0, 0]).all(): adequeate_adequeate.append(count)\n    elif (i == [-1, 1, 0]).all(): adequeate_ineffective.append(count)\n    elif (i == [0, 1, -1]).all(): adequeate_effective.append(count)\n    count = count+1\n\ncount = 0\nfor i in label[label_dict[\"effective\"]] - output_score[label_dict[\"effective\"]]:\n    if (i == [0, 0, 0]).all(): effective_effective.append(count)\n    elif (i == [-1, 0, 1]).all(): effective_ineffective.append(count)\n    elif (i == [0, -1, 1]).all(): effective_adequeate.append(count)\n    count = count+1","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:16:46.315890Z","iopub.execute_input":"2022-08-23T15:16:46.316244Z","iopub.status.idle":"2022-08-23T15:16:46.367619Z","shell.execute_reply.started":"2022-08-23T15:16:46.316214Z","shell.execute_reply":"2022-08-23T15:16:46.366686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"ineffective:\",label[label_dict[\"ineffective\"]].shape[0])\nprint(\"ineffective-ineffective:\",\"{:.2f}%\".format(len(ineffective_ineffective)*100/label[label_dict[\"ineffective\"]].shape[0]))\nprint(\"ineffective-adequeate:\",\"{:.2f}%\".format(len(ineffective_adequeate)*100/label[label_dict[\"ineffective\"]].shape[0]))\nprint(\"ineffective-effective:\",\"{:.2f}%\".format(len(ineffective_effective)*100/label[label_dict[\"ineffective\"]].shape[0]),\"\\n\")\n\nprint(\"adequeate:\",label[label_dict[\"adequeate\"]].shape[0])\nprint(\"adequeate-ineffective:\",\"{:.2f}%\".format(len(adequeate_ineffective)*100/label[label_dict[\"adequeate\"]].shape[0]))\nprint(\"adequeate-adequeate:\",\"{:.2f}%\".format(len(adequeate_adequeate)*100/label[label_dict[\"adequeate\"]].shape[0]))\nprint(\"adequeate-effective:\",\"{:.2f}%\".format(len(adequeate_effective)*100/label[label_dict[\"adequeate\"]].shape[0]),\"\\n\")\n\nprint(\"effective:\",label[label_dict[\"effective\"]].shape[0])\nprint(\"effective-ineffective:\",\"{:.2f}%\".format(len(effective_ineffective)*100/label[label_dict[\"effective\"]].shape[0]))\nprint(\"effective-adequeate:\",\"{:.2f}%\".format(len(effective_adequeate)*100/label[label_dict[\"effective\"]].shape[0]))\nprint(\"effective-effective:\",\"{:.2f}%\".format(len(effective_effective)*100/label[label_dict[\"effective\"]].shape[0]),\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:16:48.811858Z","iopub.execute_input":"2022-08-23T15:16:48.812226Z","iopub.status.idle":"2022-08-23T15:16:48.827462Z","shell.execute_reply.started":"2022-08-23T15:16:48.812196Z","shell.execute_reply":"2022-08-23T15:16:48.826336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"add_essay(path=\"../input/feedback-prize-effectiveness/test.csv\", \n          txt_folder=\"../input/feedback-prize-effectiveness/test\", split_train=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:01:53.425010Z","iopub.execute_input":"2022-08-23T15:01:53.425383Z","iopub.status.idle":"2022-08-23T15:01:53.473308Z","shell.execute_reply.started":"2022-08-23T15:01:53.425349Z","shell.execute_reply":"2022-08-23T15:01:53.472305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_dataset = data_pipeline(\"df.csv\", set_label=False, max_len=512, num_batch=32)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:01:59.110037Z","iopub.execute_input":"2022-08-23T15:01:59.110942Z","iopub.status.idle":"2022-08-23T15:02:20.028570Z","shell.execute_reply.started":"2022-08-23T15:01:59.110895Z","shell.execute_reply":"2022-08-23T15:02:20.027165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv(\"df.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:02:26.078194Z","iopub.execute_input":"2022-08-23T15:02:26.078794Z","iopub.status.idle":"2022-08-23T15:02:26.110703Z","shell.execute_reply.started":"2022-08-23T15:02:26.078754Z","shell.execute_reply":"2022-08-23T15:02:26.109504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in submission_dataset:\n    pass\nprint(np.array(tokenizer.convert_ids_to_tokens(i[1][0].numpy()[i[1][0].numpy()>0])),\"\\n\")\nprint(np.array(tokenizer.convert_ids_to_tokens(i[1][2].numpy()[i[1][2].numpy()>0])))","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:02:45.729885Z","iopub.execute_input":"2022-08-23T15:02:45.730261Z","iopub.status.idle":"2022-08-23T15:02:45.825512Z","shell.execute_reply.started":"2022-08-23T15:02:45.730229Z","shell.execute_reply":"2022-08-23T15:02:45.824582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def submission(submission_dataset = submission_dataset):\n    submission_ouput = []\n    discourse_id = []\n    for i in submission_dataset:\n        tmp = model.predict(i)\n        submission_ouput.append(tmp)\n        for ids in i[0].numpy():\n            discourse_id.append(ids.decode('utf-8'))\n\n    submission_ouput = np.array([j for i in submission_ouput for j in i])\n    discourse_id = np.array(discourse_id)\n\n    submission = pd.DataFrame(submission_ouput, columns=[\"Ineffective\",\"Adequate\",\"Effective\"])\n    submission.insert(loc=0, column=\"discourse_id\", value=discourse_id)\n    \n    return submission\n\nsubmission = submission(submission_dataset = submission_dataset)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:02:56.429775Z","iopub.execute_input":"2022-08-23T15:02:56.430176Z","iopub.status.idle":"2022-08-23T15:03:08.368981Z","shell.execute_reply.started":"2022-08-23T15:02:56.430143Z","shell.execute_reply":"2022-08-23T15:03:08.367867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T11:06:02.183362Z","iopub.execute_input":"2022-08-16T11:06:02.184003Z","iopub.status.idle":"2022-08-16T11:06:02.189960Z","shell.execute_reply.started":"2022-08-16T11:06:02.183963Z","shell.execute_reply":"2022-08-16T11:06:02.188951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}