{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Reference: https://www.kaggle.com/chongjiujjin/capsule-net-with-gru","metadata":{"_uuid":"01f361ddc47e0b386595316fe3d7f4dabbd260db"}},{"cell_type":"code","source":"import os\nimport time\nimport math\n\nscale = math.pi\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-25T10:47:11.00189Z","iopub.execute_input":"2023-05-25T10:47:11.002205Z","iopub.status.idle":"2023-05-25T10:47:11.793905Z","shell.execute_reply.started":"2023-05-25T10:47:11.002148Z","shell.execute_reply":"2023-05-25T10:47:11.793044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport re\nimport nltk\nfrom nltk.corpus import stopwords\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout\nfrom keras.optimizers import Adam\nfrom keras.callbacks import EarlyStopping","metadata":{"execution":{"iopub.status.busy":"2023-05-25T10:47:17.632611Z","iopub.execute_input":"2023-05-25T10:47:17.632924Z","iopub.status.idle":"2023-05-25T10:47:18.143742Z","shell.execute_reply.started":"2023-05-25T10:47:17.632868Z","shell.execute_reply":"2023-05-25T10:47:18.142857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/books1/Book1.csv\")\n#test_df = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",df.shape)\n#print(\"Test shape : \",test_df.shape)","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2023-05-25T10:47:27.480928Z","iopub.execute_input":"2023-05-25T10:47:27.481243Z","iopub.status.idle":"2023-05-25T10:47:27.538772Z","shell.execute_reply.started":"2023-05-25T10:47:27.481188Z","shell.execute_reply":"2023-05-25T10:47:27.537029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Clean the text data\ndef clean_text(text):\n    if pd.isnull(text):\n        return \"\"\n    # Remove non-alphabetic characters\n    text = re.sub('[^A-Za-z]', ' ', text)\n    # Convert to lowercase\n    text = text.lower()\n    # Tokenize the text\n    words = nltk.word_tokenize(text)\n    # Remove stop words\n    words = [word for word in words if word not in stopwords.words('english')]\n    # Join the words back into a sentence\n    text = ' '.join(words)\n    return text\n\ndf['clean_text'] = df['parent_comment'].apply(clean_text)\n\n# Split the dataset into training and testing sets","metadata":{"execution":{"iopub.status.busy":"2023-05-25T10:47:34.529795Z","iopub.execute_input":"2023-05-25T10:47:34.530117Z","iopub.status.idle":"2023-05-25T10:47:44.169906Z","shell.execute_reply.started":"2023-05-25T10:47:34.53006Z","shell.execute_reply":"2023-05-25T10:47:44.169028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data, test_data, train_labels, test_labels = train_test_split(df['clean_text'], df['label'], test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-25T10:47:50.126915Z","iopub.execute_input":"2023-05-25T10:47:50.127242Z","iopub.status.idle":"2023-05-25T10:47:50.136423Z","shell.execute_reply.started":"2023-05-25T10:47:50.127184Z","shell.execute_reply":"2023-05-25T10:47:50.135558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer","metadata":{"execution":{"iopub.status.busy":"2023-05-25T10:47:57.134607Z","iopub.execute_input":"2023-05-25T10:47:57.134926Z","iopub.status.idle":"2023-05-25T10:47:57.139298Z","shell.execute_reply.started":"2023-05-25T10:47:57.134871Z","shell.execute_reply":"2023-05-25T10:47:57.138301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  Convert text to vectors using CountVectorizer\nvectorizer = CountVectorizer()\ntrain_data_vec = vectorizer.fit_transform(train_data).toarray()\ntest_data_vec = vectorizer.transform(test_data).toarray()","metadata":{"execution":{"iopub.status.busy":"2023-05-25T10:47:57.752079Z","iopub.execute_input":"2023-05-25T10:47:57.752401Z","iopub.status.idle":"2023-05-25T10:47:57.860146Z","shell.execute_reply.started":"2023-05-25T10:47:57.752341Z","shell.execute_reply":"2023-05-25T10:47:57.859284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\n# Tokenize the texts\ntokenizer = Tokenizer()\ntokenizer.fit_on_texts(train_data)\ntrain_sequences = tokenizer.texts_to_sequences(train_data)\ntest_sequences = tokenizer.texts_to_sequences(test_data)\n# Pad the sequences\nmax_length = 100 # define the maximum length of sequences\ntrain_data_seq = pad_sequences(train_sequences, maxlen=max_length)\ntest_data_seq = pad_sequences(test_sequences, maxlen=max_length)","metadata":{"execution":{"iopub.status.busy":"2023-05-25T10:47:59.613503Z","iopub.execute_input":"2023-05-25T10:47:59.61383Z","iopub.status.idle":"2023-05-25T10:47:59.745614Z","shell.execute_reply.started":"2023-05-25T10:47:59.613775Z","shell.execute_reply":"2023-05-25T10:47:59.744644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_seq.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-25T10:48:01.01716Z","iopub.execute_input":"2023-05-25T10:48:01.017507Z","iopub.status.idle":"2023-05-25T10:48:01.024646Z","shell.execute_reply.started":"2023-05-25T10:48:01.017443Z","shell.execute_reply":"2023-05-25T10:48:01.023817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_seq.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-25T10:48:02.782104Z","iopub.execute_input":"2023-05-25T10:48:02.782642Z","iopub.status.idle":"2023-05-25T10:48:02.788925Z","shell.execute_reply.started":"2023-05-25T10:48:02.782368Z","shell.execute_reply":"2023-05-25T10:48:02.787948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-25T10:48:04.711209Z","iopub.execute_input":"2023-05-25T10:48:04.711553Z","iopub.status.idle":"2023-05-25T10:48:04.717043Z","shell.execute_reply.started":"2023-05-25T10:48:04.711493Z","shell.execute_reply":"2023-05-25T10:48:04.71628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ## split to train and val\n# train_df, val_df = train_test_split(train_df, test_size=0.1, random_state=2018)\n\n# ## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 50000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 100 # max number of words in a question to use\n\n# ## fill up the missing values\n# train_X = train_df[\"question_text\"].fillna(\"_na_\").values\n# val_X = val_df[\"question_text\"].fillna(\"_na_\").values\n# test_X = test_df[\"question_text\"].fillna(\"_na_\").values\n\n# ## Tokenize the sentences\n# tokenizer = Tokenizer(num_words=max_features)\n# tokenizer.fit_on_texts(list(train_X))\n# train_X = tokenizer.texts_to_sequences(train_X)\n# val_X = tokenizer.texts_to_sequences(val_X)\n# test_X = tokenizer.texts_to_sequences(test_X)\n\n# ## Pad the sentences \n# train_X = pad_sequences(train_X, maxlen=maxlen)\n# val_X = pad_sequences(val_X, maxlen=maxlen)\n# test_X = pad_sequences(test_X, maxlen=maxlen)\n\n# ## Get the target values\n# train_y = train_df['target'].values\n# val_y = val_df['target'].values","metadata":{"_uuid":"ba5a1b8109dee2c9fbc628d5da4a7c3447d42fb8","execution":{"iopub.status.busy":"2023-05-25T10:49:27.453295Z","iopub.execute_input":"2023-05-25T10:49:27.453662Z","iopub.status.idle":"2023-05-25T10:49:27.460782Z","shell.execute_reply.started":"2023-05-25T10:49:27.453601Z","shell.execute_reply":"2023-05-25T10:49:27.45957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"maxlen = 100","metadata":{"execution":{"iopub.status.busy":"2023-05-25T10:49:29.766941Z","iopub.execute_input":"2023-05-25T10:49:29.767262Z","iopub.status.idle":"2023-05-25T10:49:29.771091Z","shell.execute_reply.started":"2023-05-25T10:49:29.767206Z","shell.execute_reply":"2023-05-25T10:49:29.770332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##EMBEDDING_FILE1 = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n##def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n##embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE1))\n\n#all_embs = np.stack(embeddings_index.values())\n#emb_mean,emb_std = all_embs.mean(), all_embs.std()\n#embed_size = all_embs.shape[1]\n\n#word_index = tokenizer.word_index\n#nb_words = min(max_features, len(word_index))\n#embedding_matrix1 = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n#for word, i in word_index.items():\n   # if i >= max_features: continue\n    #embedding_vector = embeddings_index.get(word)\n    #if embedding_vector is not None: embedding_matrix1[i] = embedding_vector\n        ","metadata":{"_uuid":"23f130e80159bb1701e449e2e91199dbfff1f1d4","execution":{"iopub.status.busy":"2023-05-25T10:49:30.88667Z","iopub.execute_input":"2023-05-25T10:49:30.886993Z","iopub.status.idle":"2023-05-25T10:49:30.890811Z","shell.execute_reply.started":"2023-05-25T10:49:30.886937Z","shell.execute_reply":"2023-05-25T10:49:30.890006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.layers import K, Activation\nfrom keras.engine import Layer\nfrom keras.layers import Dense, Input, Embedding, Dropout, Bidirectional, GRU, Flatten, SpatialDropout1D\ngru_len = 128\nRoutings = 3\nNum_capsule = 10\nDim_capsule = 8\ndropout_p = 0.25\nrate_drop_dense = 0.28\n\ndef squash(x, axis=-1):\n    # s_squared_norm is really small\n    # s_squared_norm = K.sum(K.square(x), axis, keepdims=True) + K.epsilon()\n    # scale = K.sqrt(s_squared_norm)/ (0.5 + s_squared_norm)\n    # return scale * x\n    s_squared_norm = K.sum(K.square(x), axis, keepdims=True)\n    scale = K.sqrt(s_squared_norm + K.epsilon())\n    return x / scale\n\n\n# A Capsule Implement with Pure Keras\nclass Capsule(Layer):\n    def __init__(self, num_capsule, dim_capsule, routings=3, kernel_size=(9, 1), share_weights=True,\n                 activation='default', **kwargs):\n        super(Capsule, self).__init__(**kwargs)\n        self.num_capsule = num_capsule\n        self.dim_capsule = dim_capsule\n        self.routings = routings\n        self.kernel_size = kernel_size\n        self.share_weights = share_weights\n        if activation == 'default':\n            self.activation = squash\n        else:\n            self.activation = Activation(activation)\n\n    def build(self, input_shape):\n        super(Capsule, self).build(input_shape)\n        input_dim_capsule = input_shape[-1]\n        if self.share_weights:\n            self.W = self.add_weight(name='capsule_kernel',\n                                     shape=(1, input_dim_capsule,\n                                            self.num_capsule * self.dim_capsule),\n                                     # shape=self.kernel_size,\n                                     initializer='glorot_uniform',\n                                     trainable=True)\n        else:\n            input_num_capsule = input_shape[-2]\n            self.W = self.add_weight(name='capsule_kernel',\n                                     shape=(input_num_capsule,\n                                            input_dim_capsule,\n                                            self.num_capsule * self.dim_capsule),\n                                     initializer='glorot_uniform',\n                                     trainable=True)\n\n    def call(self, u_vecs):\n        if self.share_weights:\n            u_hat_vecs = K.conv1d(u_vecs, self.W)\n        else:\n            u_hat_vecs = K.local_conv1d(u_vecs, self.W, [1], [1])\n\n        batch_size = K.shape(u_vecs)[0]\n        input_num_capsule = K.shape(u_vecs)[1]\n        u_hat_vecs = K.reshape(u_hat_vecs, (batch_size, input_num_capsule,\n                                            self.num_capsule, self.dim_capsule))\n        u_hat_vecs = K.permute_dimensions(u_hat_vecs, (0, 2, 1, 3))\n        # final u_hat_vecs.shape = [None, num_capsule, input_num_capsule, dim_capsule]\n\n        b = K.zeros_like(u_hat_vecs[:, :, :, 0])  # shape = [None, num_capsule, input_num_capsule]\n        for i in range(self.routings):\n            b = K.permute_dimensions(b, (0, 2, 1))  # shape = [None, input_num_capsule, num_capsule]\n            c = K.softmax(b)\n            c = K.permute_dimensions(c, (0, 2, 1))\n            b = K.permute_dimensions(b, (0, 2, 1))\n            outputs = self.activation(K.batch_dot(c, u_hat_vecs, [2, 2]))\n            if i < self.routings - 1:\n                b = K.batch_dot(outputs, u_hat_vecs, [2, 3])\n\n        return outputs\n\n    def compute_output_shape(self, input_shape):\n        return (None, self.num_capsule, self.dim_capsule)\n\n\ndef get_model():\n    input1 = Input(shape=(maxlen,))\n    embed_layer = Embedding(max_features,\n                            embed_size,\n                            input_length=maxlen,\n                         #   weights=[embedding_matrix1],\n                            trainable=False)(input1)\n    embed_layer = SpatialDropout1D(rate_drop_dense)(embed_layer)\n\n    x = Bidirectional(\n        CuDNNGRU(gru_len, return_sequences=True))(\n        embed_layer)\n    capsule = Capsule(num_capsule=Num_capsule, dim_capsule=Dim_capsule, routings=Routings,\n                      share_weights=True)(x)\n    # output_capsule = Lambda(lambda x: K.sqrt(K.sum(K.square(x), 2)))(capsule)\n    capsule = Flatten()(capsule)\n    capsule = Dropout(dropout_p)(capsule)\n    output = Dense(1, activation='sigmoid')(capsule)\n    model = Model(inputs=input1, outputs=output)\n    model.compile(\n        loss='binary_crossentropy',\n        optimizer='adam',\n        metrics=['accuracy'])\n    model.summary()\n    return model","metadata":{"_uuid":"dbac46871002255165897a8969449b1d4188fd2f","execution":{"iopub.status.busy":"2023-05-25T10:49:33.334366Z","iopub.execute_input":"2023-05-25T10:49:33.334718Z","iopub.status.idle":"2023-05-25T10:49:33.364361Z","shell.execute_reply.started":"2023-05-25T10:49:33.334658Z","shell.execute_reply":"2023-05-25T10:49:33.363331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = get_model()\n\n\n","metadata":{"_uuid":"c1c51b26ae2edba9d0e3361a100ae6e268710b1a","execution":{"iopub.status.busy":"2023-05-25T10:49:34.818266Z","iopub.execute_input":"2023-05-25T10:49:34.818599Z","iopub.status.idle":"2023-05-25T10:49:44.799129Z","shell.execute_reply.started":"2023-05-25T10:49:34.818539Z","shell.execute_reply":"2023-05-25T10:49:44.798242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_X.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-25T10:50:04.332858Z","iopub.execute_input":"2023-05-25T10:50:04.333188Z","iopub.status.idle":"2023-05-25T10:50:04.336837Z","shell.execute_reply.started":"2023-05-25T10:50:04.333129Z","shell.execute_reply":"2023-05-25T10:50:04.335863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.callbacks import EarlyStopping\nearlystopping = EarlyStopping(patience=2, verbose=1, restore_best_weights=True)\nmodel.fit(train_data_seq, train_labels, batch_size=64, epochs=50, validation_data=(test_data_seq, test_labels), callbacks=[earlystopping])","metadata":{"_uuid":"81e005d6f10df82b00b506355add0a4403e23699","execution":{"iopub.status.busy":"2023-05-25T10:50:08.153789Z","iopub.execute_input":"2023-05-25T10:50:08.154129Z","iopub.status.idle":"2023-05-25T10:50:18.461316Z","shell.execute_reply.started":"2023-05-25T10:50:08.154066Z","shell.execute_reply":"2023-05-25T10:50:18.460376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build the model\nmodel = Sequential()\nmodel.add(Embedding(input_dim=len(tokenizer.word_index)+1, output_dim=100, input_length=max_length))\nmodel.add(LSTM(128))\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\n# Train the model\nearly_stopping = EarlyStopping(monitor='val_loss', patience=5, verbose=1)\nhistory = model.fit(train_data_seq, train_labels, validation_data=(test_data_seq, test_labels), epochs=15, batch_size=128, callbacks=[early_stopping])","metadata":{"execution":{"iopub.status.busy":"2023-05-25T10:53:09.539294Z","iopub.execute_input":"2023-05-25T10:53:09.539667Z","iopub.status.idle":"2023-05-25T10:53:26.293602Z","shell.execute_reply.started":"2023-05-25T10:53:09.539603Z","shell.execute_reply":"2023-05-25T10:53:26.292546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\nscore, acc = model.evaluate(test_data_seq, test_labels, verbose=0)\nprint(\"Test Accuracy: \", acc)","metadata":{"execution":{"iopub.status.busy":"2023-05-25T10:53:30.784123Z","iopub.execute_input":"2023-05-25T10:53:30.784467Z","iopub.status.idle":"2023-05-25T10:53:31.452433Z","shell.execute_reply.started":"2023-05-25T10:53:30.784392Z","shell.execute_reply":"2023-05-25T10:53:31.451258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_glove_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_glove_val_y>thresh).astype(int))))","metadata":{"_uuid":"ff43855164472de035a5a1d80b3db4838684701a","execution":{"iopub.status.busy":"2023-05-23T11:19:31.111469Z","iopub.status.idle":"2023-05-23T11:19:31.113712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Results seem to be better than the model without pretrained embeddings.","metadata":{"_uuid":"d2a33c252f31fddcc65896053184226128562776"}},{"cell_type":"code","source":"pred_glove_test_y = model.predict([test_X], batch_size=1024, verbose=1)","metadata":{"_uuid":"d51ff8ed6a87b488fec3ac84ca50df661d7c8193","execution":{"iopub.status.busy":"2023-05-23T11:19:31.1163Z","iopub.status.idle":"2023-05-23T11:19:31.116994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_test_y = (pred_glove_test_y>0.34).astype(int)\nout_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","metadata":{"_uuid":"39d4fedab4ac170863a0ee1ca3aa9be1ee58fe02","execution":{"iopub.status.busy":"2023-05-23T11:19:31.121042Z","iopub.status.idle":"2023-05-23T11:19:31.121751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"1e6323702a14c45113298eb6c6a2c7ec37e6b540","trusted":true},"execution_count":null,"outputs":[]}]}