{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nfrom tqdm.notebook import tqdm\nimport re\nfrom transformers import BertModel, BertTokenizer\nimport transformers\nfrom tokenizers import BertWordPieceTokenizer\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-21T03:04:13.107453Z","iopub.execute_input":"2022-07-21T03:04:13.107836Z","iopub.status.idle":"2022-07-21T03:04:13.118290Z","shell.execute_reply.started":"2022-07-21T03:04:13.107802Z","shell.execute_reply":"2022-07-21T03:04:13.117026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install tokenizers","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:04:13.495371Z","iopub.execute_input":"2022-07-21T03:04:13.495735Z","iopub.status.idle":"2022-07-21T03:04:13.500167Z","shell.execute_reply.started":"2022-07-21T03:04:13.495691Z","shell.execute_reply":"2022-07-21T03:04:13.498877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1- Load & Clean Data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/tweet-sentiment-extraction/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:11:06.472293Z","iopub.execute_input":"2022-07-21T03:11:06.472634Z","iopub.status.idle":"2022-07-21T03:11:06.549752Z","shell.execute_reply.started":"2022-07-21T03:11:06.472605Z","shell.execute_reply":"2022-07-21T03:11:06.548513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def basic_cleaning(text):\n    \"\"\"\n    clear url/ not alpha/ fuck-bitch swear\n    \"\"\"\n    text = re.sub(r'https?://www\\.\\S+\\.cm', '', text)\n    text = re.sub(r'[^a-zA-Z|\\s]', '', text)\n    text = re.sub(r'\\*+', 'swear', text)\n    return text\n\ndef remove_html(text):\n    html = re.compile(r'<.*?>')\n    return html.sub(r'',text)\n\ndef remove_emoji(text):\n    #emoticons\n    #symbols & pictographs\n    #transport & map symbols\n    #flags (iOS)\n    emoji_pattern = re.compile(\"[\"\\\n        u\"\\U0001F600-\\U0001F64F|\"\\\n        u\"\\U0001F300-\\U0001F5FF|\"\\\n        u\"\\U0001F680-\\U0001F6FF|\"\\\n        u\"\\U0001F1E0-\\U0001F1FF|\"\\\n        u\"\\U00002702-\\U000027B0|\"\\\n        u\"\\U000024C2-\\U0001F251\"\\\n        \"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r'', text)\n\n# remove repeated characters\ndef remove_multiplechars(text):\n    \"\"\"\n    for example, so we have “way” instead of “waaaayyyyy”\n    \"\"\"\n    text = re.sub(r'(.)\\1{3,}', r'\\1', text)\n    return text\n\ndef clean(df):\n    for col in ['text']:#,'selected_text']:\n        df[col] = df[col].astype(str).apply(lambda x:basic_cleaning(x))\n        df[col] = df[col].astype(str).apply(lambda x:remove_emoji(x))\n        df[col] = df[col].astype(str).apply(lambda x:remove_html(x))\n        df[col] = df[col].astype(str).apply(lambda x:remove_multiplechars(x))\n    return df.sample(frac=1)\n\ndf_clean = clean(df)\ndf_clean_selection = df_clean.sample(frac=1)\n# df_clean_selection = pd.concat([df_clean.sample(frac=1), df_clean[df_clean.textID.isin(resmaple_id)],\n#                                 df_clean[df_clean.textID.isin(resmaple_id)]], axis=0, ignore_index=True)\nX = df_clean_selection.text.values\ny, uniques = pd.factorize(df_clean_selection.sentiment, sort=True)\ny_tf = pd.get_dummies(df_clean_selection.sentiment)\nprint('clean Done')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:29:41.392595Z","iopub.execute_input":"2022-07-21T03:29:41.393303Z","iopub.status.idle":"2022-07-21T03:29:42.121069Z","shell.execute_reply.started":"2022-07-21T03:29:41.393266Z","shell.execute_reply":"2022-07-21T03:29:42.119714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_clean_selection.sentiment.value_counts().plot.pie()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:29:42.874663Z","iopub.execute_input":"2022-07-21T03:29:42.875338Z","iopub.status.idle":"2022-07-21T03:29:42.965498Z","shell.execute_reply.started":"2022-07-21T03:29:42.875302Z","shell.execute_reply":"2022-07-21T03:29:42.964194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2- Text encode","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:44:54.007064Z","iopub.execute_input":"2022-07-08T02:44:54.008405Z","iopub.status.idle":"2022-07-08T02:44:54.024801Z","shell.execute_reply.started":"2022-07-08T02:44:54.008362Z","shell.execute_reply":"2022-07-08T02:44:54.023506Z"}}},{"cell_type":"code","source":"# load bert pretrained model\nbert_tokenizer = transformers.AutoTokenizer.from_pretrained(\"distilbert-base-uncased\")\n# save the loaded tokenizer  locally\nsave_path = '/kaggle/working/distilbert_base_uncased'\nif not os.path.exists(save_path):\n    os.makedirs(save_path)\n\nbert_tokenizer.save_pretrained(save_path)\n# reload it with huggingface tokenizers library\nfast_tokenizer = BertWordPieceTokenizer('distilbert_base_uncased/vocab.txt', lowercase=True)\nfast_tokenizer","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:29:45.079343Z","iopub.execute_input":"2022-07-21T03:29:45.080406Z","iopub.status.idle":"2022-07-21T03:29:54.298923Z","shell.execute_reply.started":"2022-07-21T03:29:45.080367Z","shell.execute_reply":"2022-07-21T03:29:54.297959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=128):\n    \"\"\"\n    same length\n    \"\"\"\n    tokenizer.enable_truncation(max_length=maxlen)\n    tokenizer.enable_padding(length=maxlen)\n    all_ids = []\n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    return np.array(all_ids)\n\n# 该方法并没有padding\n# bert_tokenizer.encode(' Id have responded if I were going',\n#                       stride=0,\n#                       padding=True, \n#                       truncation=True, max_length=maxlen)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:29:54.300926Z","iopub.execute_input":"2022-07-21T03:29:54.301273Z","iopub.status.idle":"2022-07-21T03:29:54.310379Z","shell.execute_reply.started":"2022-07-21T03:29:54.301236Z","shell.execute_reply":"2022-07-21T03:29:54.309381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"texts = df_clean_selection.text.astype(str)\nX = fast_encode(\n    texts,\n    fast_tokenizer,\n    maxlen=128\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:29:54.311848Z","iopub.execute_input":"2022-07-21T03:29:54.312297Z","iopub.status.idle":"2022-07-21T03:29:56.621751Z","shell.execute_reply.started":"2022-07-21T03:29:54.312259Z","shell.execute_reply":"2022-07-21T03:29:56.620691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3- model","metadata":{"execution":{"iopub.status.busy":"2022-07-08T05:06:40.782424Z","iopub.execute_input":"2022-07-08T05:06:40.782897Z","iopub.status.idle":"2022-07-08T05:06:40.793637Z","shell.execute_reply.started":"2022-07-08T05:06:40.782859Z","shell.execute_reply":"2022-07-08T05:06:40.792169Z"}}},{"cell_type":"markdown","source":"## torch","metadata":{}},{"cell_type":"code","source":"import torch\nfrom torch import nn\nfrom torch.nn import functional as F\nfrom torch.utils.data import DataLoader\nfrom torch.optim import AdamW\nfrom transformers import BertModel\n\n\nclass distilBert(nn.Module):\n    def __init__(self, label_nums):\n        super(distilBert, self).__init__()\n        transformer_layer = transformers.DistilBertModel.from_pretrained('distilbert-base-uncased')\n        p = next(transformer_layer.parameters())\n#         self.embed = nn.Embedding(\n#             num_embeddings = p.shape[0], \n#             embedding_dim = p.shape[1]\n#         )\n#         self.embed.weight = p\n#         self.embed.weight.requires_grad = False\n        self.embed = nn.Embedding.from_pretrained(\n            p, freeze=True\n        )\n        self.lstm_layer = nn.LSTM(input_size=768, hidden_size=50, bidirectional=True)\n        self.lstm_layer2 = nn.LSTM(input_size=50*2, hidden_size=25, bidirectional=True)\n        self.drop1 = nn.Dropout(0.5)\n        self.fc1 = nn.Linear(50, 50)\n        self.relu = nn.ReLU(inplace=True)\n        self.drop2 = nn.Dropout(0.5)\n        self.fc2 = nn.Linear(50, label_nums)\n\n#         self.fc = nn.Sequential(\n#             nn.Dropout(0.5),\n#             nn.Linear(50, 50),\n#             nn.ReLU(inplace=True),\n#             nn.Dropout(0.5),\n        \n#             nn.Linear(50, label_nums)\n#         )\n        self._reinitialize()\n\n    def _reinitialize(self):\n        \"\"\"\n        Tensorflow/Keras-like initialization\n        \"\"\"\n        for name, p in self.named_parameters():\n            if 'lstm' in name:\n                if 'weight_ih' in name:\n                    nn.init.xavier_uniform_(p.data)\n                elif 'weight_hh' in name:\n                    nn.init.orthogonal_(p.data)\n                elif 'bias_ih' in name:\n                    p.data.fill_(0)\n                    # Set forget-gate bias to 1\n                    n = p.size(0)\n                    p.data[(n // 4):(n // 2)].fill_(1)\n                elif 'bias_hh' in name:\n                    p.data.fill_(0)\n            elif 'fc' in name:\n                if 'weight' in name:\n                    nn.init.xavier_uniform_(p.data)\n                elif 'bias' in name:\n                    p.data.fill_(0)\n\n    def forward(self, inputs):\n        out = self.embed(inputs)\n#         print('embed: ', out.shape)\n#         out = out[:,0,:].view(-1, 768)\n        out,(h,c) = self.lstm_layer(out)\n#         print('lstm_layer: ', out.shape)\n        out,(h,c) = self.lstm_layer2(out)\n#         print('lstm_layer2: ', out.shape)\n#         out = out[:,0,:].view(-1, 50)\n        out = out.max(axis=1).values\n#         print('max: ', out.shape)\n#         print(\"self.lstm_layer2:\", out.shape)\n        out=self.drop1(out)\n#         print('drop1: ', out.shape)\n        out=self.fc1(out)\n#         print('fc1: ', out.shape)\n        out=self.relu(out)\n        out=self.drop2(out)\n#         print('drop2: ', out.shape)\n        out=self.fc2(out)\n#         print('fc2: ', out.shape)\n        out = F.softmax(out, dim=1)\n        return out\n\n\n\nmodel = distilBert(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:13:52.075443Z","iopub.execute_input":"2022-07-21T03:13:52.076351Z","iopub.status.idle":"2022-07-21T03:13:54.865870Z","shell.execute_reply.started":"2022-07-21T03:13:52.076299Z","shell.execute_reply":"2022-07-21T03:13:54.864933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model\n# transformer_layer = transformers.DistilBertModel.from_pretrained('distilbert-base-uncased')\n# params_ = transformer_layer.parameters()\n# p = next(params_)\n# tfp = transformers.TFDistilBertModel.from_pretrained('distilbert-base-uncased').weights[0].numpy()\n# p, tfp","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:04:35.086765Z","iopub.execute_input":"2022-07-21T03:04:35.087328Z","iopub.status.idle":"2022-07-21T03:04:35.099475Z","shell.execute_reply.started":"2022-07-21T03:04:35.087287Z","shell.execute_reply":"2022-07-21T03:04:35.098316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test\ndd = bert_tokenizer.encode(' Id have responded if I were going',\n                      stride=0,\n                      padding=True, \n                      truncation=True, max_length=128)\n\ndd = torch.Tensor([dd]).long()\npred_ = model(dd)\n# print(dd)\npred_\ndd.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:04:35.112169Z","iopub.execute_input":"2022-07-21T03:04:35.112687Z","iopub.status.idle":"2022-07-21T03:04:35.165742Z","shell.execute_reply.started":"2022-07-21T03:04:35.112650Z","shell.execute_reply":"2022-07-21T03:04:35.164793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## tensorflow","metadata":{}},{"cell_type":"code","source":"# from tensorflow.keras.layers import (\n#     Input, Embedding, Bidirectional, LSTM,\n#     GlobalMaxPool1D, Dropout, Dense\n# )\n# from tensorflow.keras import Model\n# from tensorflow.keras.initializers import Constant\n\n# transformer_layer = transformers.TFDistilBertModel.from_pretrained('distilbert-base-uncased')\n\n# inp = Input(shape=(128, ))\n# embedding_matrix=transformer_layer.weights[0].numpy()\n# x = Embedding(embedding_matrix.shape[0],\n#             embedding_matrix.shape[1],\n#             embeddings_initializer=Constant(embedding_matrix),\n#             trainable=False)(inp)\n# x = Bidirectional(LSTM(50, return_sequences=True))(x)\n# x = Bidirectional(LSTM(25, return_sequences=True))(x)\n# x = GlobalMaxPool1D()(x)\n# x = Dropout(0.5)(x)\n# x = Dense(50, activation='relu', kernel_regularizer='L1L2')(x)\n# x = Dropout(0.5)(x)\n# x = Dense(3, activation='softmax')(x)\n\n# model_DistilBert = Model(inputs=[inp], outputs=x)\n# model_DistilBert.compile(loss='categorical_crossentropy',\n# optimizer='adam',\n# metrics=['accuracy'])\n\n# model_DistilBert.fit(\n#     X,\n#     y_tf,\n#     batch_size=32,\n#     epochs=5,\n#     validation_split=0.1\n# )","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:04:35.167198Z","iopub.execute_input":"2022-07-21T03:04:35.167778Z","iopub.status.idle":"2022-07-21T03:04:35.173758Z","shell.execute_reply.started":"2022-07-21T03:04:35.167721Z","shell.execute_reply":"2022-07-21T03:04:35.172695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_DistilBert.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:04:35.175599Z","iopub.execute_input":"2022-07-21T03:04:35.176430Z","iopub.status.idle":"2022-07-21T03:04:35.185522Z","shell.execute_reply.started":"2022-07-21T03:04:35.176383Z","shell.execute_reply":"2022-07-21T03:04:35.184221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pred_tf = model_DistilBert.predict(X)\n\n# np.argmax(pred_tf, axis=1)[:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:04:35.187395Z","iopub.execute_input":"2022-07-21T03:04:35.188038Z","iopub.status.idle":"2022-07-21T03:04:35.193518Z","shell.execute_reply.started":"2022-07-21T03:04:35.188000Z","shell.execute_reply":"2022-07-21T03:04:35.192186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4- Train","metadata":{}},{"cell_type":"code","source":"from torch.utils.data import DataLoader, TensorDataset, Dataset\nfrom torch.optim import AdamW, Adam\nimport torch\nfrom sklearn.metrics import accuracy_score\n\nb=torch.cuda.is_available()\nif(b):\n    device = torch.device(\"cuda\")\nelse:\n    device = torch.device(\"cpu\")\n\nmodel.to(device)\noptim = Adam(model.parameters(), lr=0.005)\nloss_fn = nn.CrossEntropyLoss()\n\ndef train(train_loader, epoches):\n#     optim = AdamW(model.parameters(), lr=0.001)\n    model.train()\n    total_train_loss = 0\n    iter_num = 0\n    for ep in range(epoches):\n        cnt = 0\n        loss_tt = 0\n        right_cnt = 0\n        samples_ = 0\n        for x, y in tqdm(train_loader):\n            x = x.to(device)\n            y = y.to(device)\n            cnt += 1\n            samples_ += len(y)\n            # 正向传播\n            optim.zero_grad()\n            pred_ =  model(x)\n            loss = loss_fn(pred_, y)\n            \n            loss_tt += loss\n            pred_ = pred_.cpu().detach().numpy() \n            right_cnt += np.sum(np.argmax(pred_, axis=1) == y.cpu().detach().numpy() )\n            # 反向梯度信息\n            loss.backward()\n            torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0)\n            # 参数更新\n            optim.step()\n            iter_num += 1\n            if(iter_num % 100 == 0):\n                acc_ = accuracy_score(np.argmax(pred_, axis=1), y.cpu().detach().numpy())\n                print(f\"[ iter_num-{iter_num} ]: loss: {loss:.5f}, acc: {acc_:.3f}\")\n\n        loss_tt /= cnt\n        acc_ = right_cnt / samples_\n        print(f\"[ ep: {ep} ] loss_tt: {loss_tt:.5f}, acc: {acc_:.3f}\")\n\ndef validation(val_dataloader):\n    model.eval()\n    cnt = 0\n    loss_tt = 0\n    right_cnt = 0\n    samples_ = 0\n    for x, y in tqdm(val_dataloader):\n        x = x.to(device)\n        y = y.to(device)\n        cnt += 1\n        samples_ += len(y)\n        with torch.no_grad():\n            pred_ =  model(x)\n            loss = loss_fn(pred_, y)\n        \n        pred_ = pred_.cpu().detach().numpy()\n        loss_tt += loss\n        right_cnt += np.sum(np.argmax(pred_, axis=1) == y.cpu().detach().numpy() )\n\n    loss_tt /= cnt\n    acc_ = right_cnt / samples_\n    print(\"-------------------------------\")\n    print(f\"loss_tt: {loss_tt:.5f}, acc: {acc_:.3f}\")\n    print(\"-------------------------------\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:13:59.775670Z","iopub.execute_input":"2022-07-21T03:13:59.776051Z","iopub.status.idle":"2022-07-21T03:13:59.820918Z","shell.execute_reply.started":"2022-07-21T03:13:59.776019Z","shell.execute_reply":"2022-07-21T03:13:59.819994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedShuffleSplit\n\nclass myDataset(Dataset):\n    def __init__(self, encodings, labels):\n        self.encodings = encodings\n        self.labels = labels\n    \n    # 读取单个样本\n    def __getitem__(self, idx):\n        x = torch.tensor(self.encodings[idx])\n        y = torch.tensor(int(self.labels[idx]))\n        return x, y\n    \n    def __len__(self):\n        return len(self.labels)\n\n# tr_data = myDataset(X[:-1000,:], y[:-1000])\n# val_data = myDataset(X[-1000:,:], y[-1000:])\ntr_data = TensorDataset(\n    torch.tensor(X[:-1000,:]).long(), \n    torch.tensor(y[:-1000]).long()\n)\n# tr_data = TensorDataset(\n#     torch.tensor(X[:2,:]).long(), \n#     torch.tensor(y[:2]).long()\n# )\nval_data = TensorDataset(\n    torch.tensor(X[-1000:,:]).long(), \n    torch.tensor(y[-1000:]).long()\n)\n\ntrain_loader = DataLoader(tr_data, batch_size=32, shuffle=True)\nval_dataloader = DataLoader(val_data, batch_size=32, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:14:03.638493Z","iopub.execute_input":"2022-07-21T03:14:03.638865Z","iopub.status.idle":"2022-07-21T03:14:03.657857Z","shell.execute_reply.started":"2022-07-21T03:14:03.638833Z","shell.execute_reply":"2022-07-21T03:14:03.656918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X[:2,:].shape","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:04:35.313000Z","iopub.execute_input":"2022-07-21T03:04:35.314594Z","iopub.status.idle":"2022-07-21T03:04:35.321385Z","shell.execute_reply.started":"2022-07-21T03:04:35.314556Z","shell.execute_reply":"2022-07-21T03:04:35.320180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train(train_loader=train_loader, epoches=5)\n# validation(val_dataloader=val_dataloader)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:04:35.323366Z","iopub.execute_input":"2022-07-21T03:04:35.324611Z","iopub.status.idle":"2022-07-21T03:04:35.330641Z","shell.execute_reply.started":"2022-07-21T03:04:35.324571Z","shell.execute_reply":"2022-07-21T03:04:35.329683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# pytorch use like keras","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport torch\nfrom torch import nn\nfrom torch.nn import functional as F\nimport transformers\nfrom torch.utils.data import DataLoader, TensorDataset, Dataset\nfrom torch.optim import AdamW, Adam\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import StratifiedShuffleSplit\nfrom transformers import BertModel\nimport typing  as typ\n\n\nclass distilBert(nn.Module):\n    def __init__(self, label_nums):\n        super(distilBert, self).__init__()\n        transformer_layer = transformers.DistilBertModel.from_pretrained('distilbert-base-uncased')\n        p = next(transformer_layer.parameters())\n        self.embed = nn.Embedding.from_pretrained(\n            p, freeze=True\n        )\n        self.lstm_layer = nn.LSTM(input_size=768, hidden_size=64, bidirectional=True)\n        self.lstm_layer2 = nn.LSTM(input_size=64*2, hidden_size=25, bidirectional=True)\n        self.drop1 = nn.Dropout(0.5)\n        self.fc1 = nn.Linear(50, 50)\n        self.relu = nn.LeakyReLU(inplace=True)\n        self.drop2 = nn.Dropout(0.5)\n        self.fc2 = nn.Linear(50, label_nums)\n        self._reinitialize()\n\n    def _reinitialize(self):\n        \"\"\"\n        Tensorflow/Keras-like initialization\n        \"\"\"\n        self.device = torch.device(\"cpu\")\n        for name, p in self.named_parameters():\n            if 'lstm' in name:\n                if 'weight_ih' in name:\n                    nn.init.xavier_uniform_(p.data)\n                elif 'weight_hh' in name:\n                    nn.init.orthogonal_(p.data)\n                elif 'bias_ih' in name:\n                    p.data.fill_(0)\n                    # Set forget-gate bias to 1\n                    n = p.size(0)\n                    p.data[(n // 4):(n // 2)].fill_(1)\n                elif 'bias_hh' in name:\n                    p.data.fill_(0)\n            elif 'fc' in name:\n                if 'weight' in name:\n                    nn.init.xavier_uniform_(p.data)\n                elif 'bias' in name:\n                    p.data.fill_(0)\n\n    def forward(self, inputs):\n        out = self.embed(inputs)\n        out,(h,c) = self.lstm_layer(out)\n        out,(h,c) = self.lstm_layer2(out)\n        out = out.max(axis=1).values\n        out=self.drop1(out)\n        out=self.fc1(out)\n        out=self.relu(out)\n        out=self.drop2(out)\n        out=self.fc2(out)\n        out = F.softmax(out, dim=1)\n        return out\n    \n    def compile(self, \n                loss: typ.Callable,\n                optimizer: typ.Callable,\n                learning_rate: float,\n                metrics: typ.List[typ.Callable],\n                verbose: int=100):\n        self.verbose = verbose\n        b=torch.cuda.is_available()\n        if(b):\n            self.device = torch.device(\"cuda\")\n        else:\n            self.device = torch.device(\"cpu\")\n\n        self.to(self.device)\n        self.loss_fn = loss\n        self.optim = optimizer(self.parameters(), lr=learning_rate)\n        self.metrics = metrics\n        self.iter_num = 0\n\n    def _training_step(self, x, y):\n        x = x.to(self.device)\n        y = y.to(self.device)\n        self.epoch_steps += 1\n        self.epoch_samples += len(y)\n        # 正向传播\n        self.optim.zero_grad()\n        pred_ =  self.forward(x)\n        loss = self.loss_fn(pred_, y)\n        \n        pred_ = pred_.cpu().detach().numpy() \n        self.epoch_loss += loss\n        self.epoch_right_samples += np.sum(np.argmax(pred_, axis=1) == y.cpu().detach().numpy() )\n        # 反向梯度信息\n        loss.backward()\n        torch.nn.utils.clip_grad_norm_(self.parameters(), 1.0)\n        # 参数更新\n        self.optim.step()\n        self.iter_num += 1\n        if(self.iter_num % self.verbose == 0):\n            metric_str = ''\n            for metric_fn in self.metrics:\n                metric_res = metric_fn(np.argmax(pred_, axis=1), y.cpu().detach().numpy())\n                metric_str += f'{metric_fn.__name__} : {metric_res:.3f}, '\n\n            print(f\"[ iter_num-{self.iter_num} ]: loss: {loss:.5f}  \" + metric_str)\n    \n    def train_one_epoch(self, dataloader, ep_idx=0):\n        self.epoch_loss = 0\n        self.epoch_samples = 0\n        self.epoch_right_samples = 0\n        self.epoch_steps = 0\n        for x, y in tqdm(dataloader):\n            self._training_step(x, y)\n        \n        self.epoch_loss /= self.epoch_steps\n        acc_ = self.epoch_right_samples / self.epoch_samples\n        print(f\"[ epoch: {ep_idx} ] loss_tt: {self.epoch_loss:.5f}, acc: {acc_:.3f}\")\n\n    def _valitate_step(self, x, y):\n        x = x.to(self.device)\n        y = y.to(self.device)\n        self.epoch_steps += 1\n        self.epoch_samples += len(y)\n        pred_ =  self.forward(x)\n        loss = self.loss_fn(pred_, y)\n        \n        pred_ = pred_.cpu().detach().numpy() \n        self.epoch_loss += loss\n        self.epoch_right_samples += np.sum(np.argmax(pred_, axis=1) == y.cpu().detach().numpy() )\n\n        self.iter_num += 1\n        if(self.iter_num % self.verbose == 0):\n            metric_str = ''\n            for metric_fn in self.metrics:\n                metric_res = metric_fn(np.argmax(pred_, axis=1), y.cpu().detach().numpy())\n                metric_str += f'{metric_fn.__name__} : {metric_res:.3f}, '\n\n            print(f\"[ iter_num-{self.iter_num} ]: loss: {loss:.5f}  \" + metric_str)\n\n    def valitate_one_epoch(self, dataloader, ep_idx=0):\n        self.epoch_loss = 0\n        self.epoch_samples = 0\n        self.epoch_right_samples = 0\n        self.epoch_steps = 0\n        for x, y in tqdm(dataloader):\n            self._valitate_step(x, y)\n\n        self.epoch_loss /= self.epoch_steps\n        acc_ = self.epoch_right_samples / self.epoch_samples\n        print(\"-------------------------------\")\n        print(f\"[ epoch: {ep_idx}-valitation ] loss_tt: {self.epoch_loss:.5f}, acc: {acc_:.3f}\")\n        print(\"-------------------------------\")\n\n    def fit(self, X, y,\n            batch_size: int=32,\n            epochs: int=5,\n            validation_split: float=0.1\n            ):\n\n        sp = StratifiedShuffleSplit(n_splits=epochs, test_size=int(len(y)*validation_split))\n        for idx, (tr_idx, te_idx) in  enumerate(sp.split(X, y)):\n            tr_data = TensorDataset(\n                torch.tensor(X[tr_idx, :]).long(), \n                torch.tensor(y[tr_idx]).long()\n            )\n\n            val_data = TensorDataset(\n                torch.tensor(X[te_idx, :]).long(), \n                torch.tensor(y[te_idx]).long()\n            )\n            train_loader = DataLoader(tr_data, batch_size=batch_size, shuffle=True)\n            val_dataloader = DataLoader(val_data, batch_size=batch_size, shuffle=True)\n            self.train_one_epoch(train_loader, idx+1)\n            self.valitate_one_epoch(val_dataloader, idx+1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T04:43:17.728851Z","iopub.execute_input":"2022-07-21T04:43:17.729263Z","iopub.status.idle":"2022-07-21T04:43:17.768906Z","shell.execute_reply.started":"2022-07-21T04:43:17.729231Z","shell.execute_reply":"2022-07-21T04:43:17.767707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = distilBert(3)\n\nmodel.compile(\n    loss=nn.CrossEntropyLoss(),\n    optimizer=AdamW,\n    learning_rate=0.0015,\n    metrics=[accuracy_score],\n    verbose=100\n)\n\nmodel.fit( X, y,\n        batch_size=32,\n        epochs=10,\n        validation_split=0.1\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T04:43:20.927186Z","iopub.execute_input":"2022-07-21T04:43:20.927532Z","iopub.status.idle":"2022-07-21T04:45:15.399159Z","shell.execute_reply.started":"2022-07-21T04:43:20.927503Z","shell.execute_reply":"2022-07-21T04:45:15.398111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bad case ana","metadata":{}},{"cell_type":"code","source":"\ndel train_loader\ndel val_dataloader\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:05:56.287548Z","iopub.execute_input":"2022-07-21T03:05:56.288659Z","iopub.status.idle":"2022-07-21T03:05:56.293984Z","shell.execute_reply.started":"2022-07-21T03:05:56.288618Z","shell.execute_reply":"2022-07-21T03:05:56.292783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32\ntr_data = TensorDataset(\n    torch.tensor(X).long(), \n    torch.tensor(y).long()\n)\n\ntrain_loader = DataLoader(tr_data, batch_size=batch_size, shuffle=False)\n\npred_list = []\nfor xi, yi in tqdm(train_loader):\n    model.eval()\n    pred_ = model.forward(xi.to(device))\n    pred_list.append(np.argmax(pred_.detach().cpu().numpy(), axis=1))\n\npred_list[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T04:45:19.461503Z","iopub.execute_input":"2022-07-21T04:45:19.462487Z","iopub.status.idle":"2022-07-21T04:45:22.367093Z","shell.execute_reply.started":"2022-07-21T04:45:19.462436Z","shell.execute_reply":"2022-07-21T04:45:22.366231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_f = np.concatenate(pred_list)\n# pred_f = np.argmax(pred_tf, axis=1)\npred_f[:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T04:45:23.660248Z","iopub.execute_input":"2022-07-21T04:45:23.660589Z","iopub.status.idle":"2022-07-21T04:45:23.670049Z","shell.execute_reply.started":"2022-07-21T04:45:23.660558Z","shell.execute_reply":"2022-07-21T04:45:23.668855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_clean_selection['pred'] = pred_f\ndf_clean_selection['pred_name'] = df_clean_selection['pred'].map(dict(zip(range(3), uniques.tolist())))\nf_bool = df_clean_selection.pred_name != df_clean_selection.sentiment\nprint(f_bool.value_counts(),'\\n ACC' ,(~f_bool).sum() / f_bool.shape[0])\ndf_clean_selection[f_bool]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T04:48:29.420099Z","iopub.execute_input":"2022-07-21T04:48:29.420448Z","iopub.status.idle":"2022-07-21T04:48:29.450530Z","shell.execute_reply.started":"2022-07-21T04:48:29.420419Z","shell.execute_reply":"2022-07-21T04:48:29.449614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_clean_selection[f_bool].sentiment.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T03:09:37.598445Z","iopub.execute_input":"2022-07-21T03:09:37.599043Z","iopub.status.idle":"2022-07-21T03:09:37.613404Z","shell.execute_reply.started":"2022-07-21T03:09:37.599005Z","shell.execute_reply":"2022-07-21T03:09:37.612564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}