{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T10:16:43.568858Z","iopub.execute_input":"2022-07-12T10:16:43.569564Z","iopub.status.idle":"2022-07-12T10:16:44.924876Z","shell.execute_reply.started":"2022-07-12T10:16:43.569465Z","shell.execute_reply":"2022-07-12T10:16:44.923506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport copy\nimport time\nimport random\nimport string\nimport joblib\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom tqdm import tqdm\nfrom collections import defaultdict\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import GroupKFold, KFold\nfrom transformers import AutoTokenizer, AutoModel, AutoConfig, AdamW\nfrom transformers import DataCollatorWithPadding\nimport warnings\n\nwarnings.filterwarnings(\"ignore\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:16:44.926788Z","iopub.execute_input":"2022-07-12T10:16:44.927127Z","iopub.status.idle":"2022-07-12T10:16:52.220395Z","shell.execute_reply.started":"2022-07-12T10:16:44.927090Z","shell.execute_reply":"2022-07-12T10:16:52.219500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#训练pipeline\ntrain_dir = \"../input/feedback-prize-effectiveness/train\"\ntest_dir = \"../input/feedback-prize-effectiveness/test\"","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:16:52.221664Z","iopub.execute_input":"2022-07-12T10:16:52.222278Z","iopub.status.idle":"2022-07-12T10:16:52.228657Z","shell.execute_reply.started":"2022-07-12T10:16:52.222239Z","shell.execute_reply":"2022-07-12T10:16:52.227777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_path = '../input/deberta-v3-base/deberta-v3-base'\n#pretrained模型预训练权重","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:16:52.231330Z","iopub.execute_input":"2022-07-12T10:16:52.232247Z","iopub.status.idle":"2022-07-12T10:16:52.239386Z","shell.execute_reply.started":"2022-07-12T10:16:52.232210Z","shell.execute_reply":"2022-07-12T10:16:52.238298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#定义所需要的config\nconfig = {\n    'seed':2022,\n    'epochs':3,\n    'model_name':\"deberta-v3-base\",\n    'train_batc_size':8,\n    'valid_batch_size':8,\n    'max_length':1024,\n    'learning_rate':2e-5,\n    'scheduler':'CosineAnnealingLR',\n    'min_lr':1e-6,\n    'T_max':500,\n    'weight_decay':1e-6,\n    'n_fold':5,\n    'n_accumulate':1,\n    'num_classes':3,\n    \"device\": torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\"),\n    \"competition\": \"FeedBack\",\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:16:52.242900Z","iopub.execute_input":"2022-07-12T10:16:52.243186Z","iopub.status.idle":"2022-07-12T10:16:52.306683Z","shell.execute_reply.started":"2022-07-12T10:16:52.243161Z","shell.execute_reply":"2022-07-12T10:16:52.305488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config['tokenizer'] = AutoTokenizer.from_pretrained(model_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:16:52.308361Z","iopub.execute_input":"2022-07-12T10:16:52.309253Z","iopub.status.idle":"2022-07-12T10:16:53.009862Z","shell.execute_reply.started":"2022-07-12T10:16:52.309211Z","shell.execute_reply":"2022-07-12T10:16:53.008918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 设置随机种子\ndef set_seed(seed=42):\n    '''Sets the seed of the entire notebook so results are the same every time we run.\n    This is for REPRODUCIBILITY.'''\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    # When running on the CuDNN backend, two further options must be set\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    # Set a fixed value for the hash seed\n    os.environ['PYTHONHASHSEED'] = str(seed)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:16:53.011376Z","iopub.execute_input":"2022-07-12T10:16:53.011760Z","iopub.status.idle":"2022-07-12T10:16:53.018315Z","shell.execute_reply.started":"2022-07-12T10:16:53.011723Z","shell.execute_reply":"2022-07-12T10:16:53.016871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_seed(config['seed'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:16:53.020003Z","iopub.execute_input":"2022-07-12T10:16:53.020816Z","iopub.status.idle":"2022-07-12T10:16:53.032079Z","shell.execute_reply.started":"2022-07-12T10:16:53.020703Z","shell.execute_reply":"2022-07-12T10:16:53.031111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_essay(essay_id):\n    essay_path = os.path.join(train_dir,f'{essay_id}.txt')\n    essay_text = open(essay_path,'r').read()\n    return essay_text","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:16:53.033689Z","iopub.execute_input":"2022-07-12T10:16:53.034353Z","iopub.status.idle":"2022-07-12T10:16:53.040303Z","shell.execute_reply.started":"2022-07-12T10:16:53.034316Z","shell.execute_reply":"2022-07-12T10:16:53.039377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/feedback-prize-effectiveness/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:16:53.044763Z","iopub.execute_input":"2022-07-12T10:16:53.045019Z","iopub.status.idle":"2022-07-12T10:16:53.312584Z","shell.execute_reply.started":"2022-07-12T10:16:53.044995Z","shell.execute_reply":"2022-07-12T10:16:53.311622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['essay_text'] = df['essay_id'].apply(get_essay)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:16:53.314056Z","iopub.execute_input":"2022-07-12T10:16:53.314409Z","iopub.status.idle":"2022-07-12T10:17:15.831408Z","shell.execute_reply.started":"2022-07-12T10:16:53.314373Z","shell.execute_reply":"2022-07-12T10:17:15.830499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:17:15.833292Z","iopub.execute_input":"2022-07-12T10:17:15.834162Z","iopub.status.idle":"2022-07-12T10:17:15.853562Z","shell.execute_reply.started":"2022-07-12T10:17:15.834120Z","shell.execute_reply":"2022-07-12T10:17:15.852722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=config['n_fold'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:17:15.856270Z","iopub.execute_input":"2022-07-12T10:17:15.856556Z","iopub.status.idle":"2022-07-12T10:17:15.862088Z","shell.execute_reply.started":"2022-07-12T10:17:15.856530Z","shell.execute_reply":"2022-07-12T10:17:15.860923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fold,(_,val_) in enumerate(gkf.split(X=df,groups=df.essay_id)):\n    df.loc[val_,'kfold'] = int(fold)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:17:15.864476Z","iopub.execute_input":"2022-07-12T10:17:15.865049Z","iopub.status.idle":"2022-07-12T10:17:15.938114Z","shell.execute_reply.started":"2022-07-12T10:17:15.865010Z","shell.execute_reply":"2022-07-12T10:17:15.937204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['kfold'] = df['kfold'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:17:15.939470Z","iopub.execute_input":"2022-07-12T10:17:15.939814Z","iopub.status.idle":"2022-07-12T10:17:15.947650Z","shell.execute_reply.started":"2022-07-12T10:17:15.939777Z","shell.execute_reply":"2022-07-12T10:17:15.946455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:17:15.949376Z","iopub.execute_input":"2022-07-12T10:17:15.949890Z","iopub.status.idle":"2022-07-12T10:17:15.968924Z","shell.execute_reply.started":"2022-07-12T10:17:15.949852Z","shell.execute_reply":"2022-07-12T10:17:15.968015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('kfold')['discourse_effectiveness'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:17:15.970260Z","iopub.execute_input":"2022-07-12T10:17:15.970666Z","iopub.status.idle":"2022-07-12T10:17:15.991195Z","shell.execute_reply.started":"2022-07-12T10:17:15.970631Z","shell.execute_reply":"2022-07-12T10:17:15.990399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder = LabelEncoder()#把三个类别变成0，1，2三个模式\ndf['discourse_effectiveness'] = encoder.fit_transform(df['discourse_effectiveness'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:19:05.523191Z","iopub.execute_input":"2022-07-12T10:19:05.523659Z","iopub.status.idle":"2022-07-12T10:19:05.541166Z","shell.execute_reply.started":"2022-07-12T10:19:05.523618Z","shell.execute_reply":"2022-07-12T10:19:05.540107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:19:08.565921Z","iopub.execute_input":"2022-07-12T10:19:08.566354Z","iopub.status.idle":"2022-07-12T10:19:08.582363Z","shell.execute_reply.started":"2022-07-12T10:19:08.566308Z","shell.execute_reply":"2022-07-12T10:19:08.580735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pkl\nwith open('le.pkl','wb') as fp:\n    joblib.dump(encoder,fp)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:20:45.684729Z","iopub.execute_input":"2022-07-12T10:20:45.685070Z","iopub.status.idle":"2022-07-12T10:20:45.690834Z","shell.execute_reply.started":"2022-07-12T10:20:45.685041Z","shell.execute_reply":"2022-07-12T10:20:45.689727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FeedBackDataset(Dataset):\n    def __init__(self,df,tokenizer,max_length):\n        self.df = df\n        self.max_len = max_length\n        self.tokenizer = tokenizer\n        self.discourse = df['discourse_text'].values\n        self.discourse_type = df['discourse_type'].values\n        self.essay = df['essay_text'].values\n        self.targets = df['discourse_effectiveness'].values\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem(self,index):\n        essay = self.essay[index]\n        discourse = self.discourse[index]\n        discourse_type = self.discourse_type[index]\n        text = dicourse_type +self.tokenizer.sep_token + discourse + self.tokenizer.sep_token + ' ' +essay\n        inputs = self.tokenizer.encode_plus(\n            text,\n            truncation=True,\n            add_special_tokens=True,\n            max_length=self.max_len,\n            padding = 'max_length'\n        )\n        return {\n            'inputs_ids':inputs['input_ids'],\n            'attention_mask':inputs['attention_mask'],\n            'target':self.targets[index]\n        }","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:46:40.987050Z","iopub.execute_input":"2022-07-12T10:46:40.987526Z","iopub.status.idle":"2022-07-12T10:46:41.004618Z","shell.execute_reply.started":"2022-07-12T10:46:40.987486Z","shell.execute_reply":"2022-07-12T10:46:41.003366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"collate_fn = DataCollatorWithPadding(tokenizer=config['tokenizer'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T10:47:44.425021Z","iopub.execute_input":"2022-07-12T10:47:44.425362Z","iopub.status.idle":"2022-07-12T10:47:44.430584Z","shell.execute_reply.started":"2022-07-12T10:47:44.425335Z","shell.execute_reply":"2022-07-12T10:47:44.429388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#MeanPooling_trick\nclass MeanPooling(nn.Module):\n    def __init(self):\n        super(MeanPooling,self).__init__()\n        \n    def forward(self,last_hidden_state,attention_mask):\n        input_mask_expand = attention_mask.unsqueeze(-1),expand(last_hidden_state.size()).float()\n        sum_embeddings = torch.sum(last_hidden_state * input_mask_expanded,1)\n        sum_mask = input_mask_expanded.sun(1)\n        sum_mask = torch.clamp(sum_mask,min=1e-9)\n        mean_embeddings = sum_embeddings/sum_mask\n        return mean_embeddings","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#构建模型，用上这个trick\nclass FeedBackModel(nn.Module):\n    def __init__(self,model_name):\n        super(FeedBackModel,self).__init__()\n        self.model = AutoModel.from_pretrained(model_name)\n        self.config = AutoConfig.from_pretrained(model_name)\n        self.drop = nn.Dropout(p=0.2)\n        self.pooler = MeanPooling()\n        self.fc = nn.Linear(self.config.hidden_size, CONFIG['num_classes'])\n        \n    def forward(self,ids,mask):\n        out = self.model(inputs_ids = ids,attention_mask=mask,output_hidden_states=False)\n        \n        ","metadata":{},"execution_count":null,"outputs":[]}]}