{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T13:32:22.284498Z","iopub.execute_input":"2022-08-13T13:32:22.284948Z","iopub.status.idle":"2022-08-13T13:32:22.322018Z","shell.execute_reply.started":"2022-08-13T13:32:22.284864Z","shell.execute_reply":"2022-08-13T13:32:22.321012Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Hello Kagglers, this is a work in progress. For now we have basic preprocessing, glove and baseline model.**","metadata":{}},{"cell_type":"markdown","source":"**TODO**\n\n* Text: Lemmatize, stem\n* Vectorization: Try out different techniques\n* Dimentionality reduction \n* Experiment with more models","metadata":{}},{"cell_type":"markdown","source":"#### **Importing libraries**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nfrom nltk.corpus import stopwords\nfrom nltk.util import ngrams\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom collections import defaultdict\nfrom collections import  Counter\nplt.style.use('ggplot')\nstop=set(stopwords.words('english'))\nimport re\nfrom nltk.tokenize import word_tokenize\nimport gensim\nimport string\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom tqdm import tqdm\nfrom keras.models import Sequential\nfrom keras.layers import Embedding,LSTM,Dense,SpatialDropout1D\nfrom keras.initializers import Constant\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:22.348108Z","iopub.execute_input":"2022-08-13T13:32:22.348536Z","iopub.status.idle":"2022-08-13T13:32:28.589701Z","shell.execute_reply.started":"2022-08-13T13:32:22.348508Z","shell.execute_reply":"2022-08-13T13:32:28.588756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **Ingesting data**","metadata":{}},{"cell_type":"code","source":"tweet = pd.read_csv('/kaggle/input/nlp-getting-started/train.csv')\ntest = pd.read_csv('/kaggle/input/nlp-getting-started/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.591558Z","iopub.execute_input":"2022-08-13T13:32:28.592264Z","iopub.status.idle":"2022-08-13T13:32:28.658533Z","shell.execute_reply.started":"2022-08-13T13:32:28.592217Z","shell.execute_reply":"2022-08-13T13:32:28.657632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"The shape of train data is: {}\".format(tweet.shape))\nprint(\"The shape of test data is: {}\".format(test.shape))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.659802Z","iopub.execute_input":"2022-08-13T13:32:28.660154Z","iopub.status.idle":"2022-08-13T13:32:28.666841Z","shell.execute_reply.started":"2022-08-13T13:32:28.660119Z","shell.execute_reply":"2022-08-13T13:32:28.665748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.669661Z","iopub.execute_input":"2022-08-13T13:32:28.670680Z","iopub.status.idle":"2022-08-13T13:32:28.690238Z","shell.execute_reply.started":"2022-08-13T13:32:28.670643Z","shell.execute_reply":"2022-08-13T13:32:28.689116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dealing with missing values","metadata":{}},{"cell_type":"code","source":"tweet.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.691516Z","iopub.execute_input":"2022-08-13T13:32:28.692376Z","iopub.status.idle":"2022-08-13T13:32:28.702952Z","shell.execute_reply.started":"2022-08-13T13:32:28.692338Z","shell.execute_reply":"2022-08-13T13:32:28.701523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Finding which is the most common keyword when referring to disaster in order to impute those few missing keyword records.**","metadata":{}},{"cell_type":"code","source":"tweet[tweet['target']==1]['keyword'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.704221Z","iopub.execute_input":"2022-08-13T13:32:28.705103Z","iopub.status.idle":"2022-08-13T13:32:28.723599Z","shell.execute_reply.started":"2022-08-13T13:32:28.705070Z","shell.execute_reply":"2022-08-13T13:32:28.722736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet[tweet['target']==0]['keyword'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.724814Z","iopub.execute_input":"2022-08-13T13:32:28.725629Z","iopub.status.idle":"2022-08-13T13:32:28.735934Z","shell.execute_reply.started":"2022-08-13T13:32:28.725596Z","shell.execute_reply":"2022-08-13T13:32:28.734834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Imputing NAs with the most common**","metadata":{}},{"cell_type":"code","source":"tweet['keyword'] = tweet.apply(\n    lambda x: 'body%20bags' if x['target']==0 else 'derailment', axis=1\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.737729Z","iopub.execute_input":"2022-08-13T13:32:28.738361Z","iopub.status.idle":"2022-08-13T13:32:28.813262Z","shell.execute_reply.started":"2022-08-13T13:32:28.738326Z","shell.execute_reply":"2022-08-13T13:32:28.812330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**For now, I am dropping the location column due to high number of NA's, I ll see if we can use it later**","metadata":{}},{"cell_type":"code","source":"tweet.drop(columns=['location'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.814763Z","iopub.execute_input":"2022-08-13T13:32:28.815107Z","iopub.status.idle":"2022-08-13T13:32:28.823911Z","shell.execute_reply.started":"2022-08-13T13:32:28.815073Z","shell.execute_reply":"2022-08-13T13:32:28.822958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.827914Z","iopub.execute_input":"2022-08-13T13:32:28.828613Z","iopub.status.idle":"2022-08-13T13:32:28.839424Z","shell.execute_reply.started":"2022-08-13T13:32:28.828579Z","shell.execute_reply":"2022-08-13T13:32:28.838348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now that we have got rid of our NAs, it's time to tackle the text. \n\nIn every text schema task I usually, clean the data by removing special characters, punctuations, numbers etc, and then apply some preprocessing in order to fit the data on our models","metadata":{}},{"cell_type":"markdown","source":"#### **Preprocessing**","metadata":{}},{"cell_type":"markdown","source":"**Concat train and test to apply same preprocess once**","metadata":{}},{"cell_type":"code","source":"df = pd.concat([tweet,test])\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.841929Z","iopub.execute_input":"2022-08-13T13:32:28.842633Z","iopub.status.idle":"2022-08-13T13:32:28.853570Z","shell.execute_reply.started":"2022-08-13T13:32:28.842598Z","shell.execute_reply":"2022-08-13T13:32:28.852629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Removing URLs**","metadata":{}},{"cell_type":"code","source":"def remove_URL(text):\n    url = re.compile(r'https?://\\S+|www\\.\\S+')\n    return url.sub(r'',text)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.855301Z","iopub.execute_input":"2022-08-13T13:32:28.855989Z","iopub.status.idle":"2022-08-13T13:32:28.860811Z","shell.execute_reply.started":"2022-08-13T13:32:28.855953Z","shell.execute_reply":"2022-08-13T13:32:28.859780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text'] = df['text'].apply(lambda x: remove_URL(x))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.861678Z","iopub.execute_input":"2022-08-13T13:32:28.862530Z","iopub.status.idle":"2022-08-13T13:32:28.893529Z","shell.execute_reply.started":"2022-08-13T13:32:28.862493Z","shell.execute_reply":"2022-08-13T13:32:28.892722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Removing HTML tags**","metadata":{}},{"cell_type":"code","source":"def remove_html(text):\n    html=re.compile(r'<.*?>')\n    return html.sub(r'',text)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.894887Z","iopub.execute_input":"2022-08-13T13:32:28.895483Z","iopub.status.idle":"2022-08-13T13:32:28.900168Z","shell.execute_reply.started":"2022-08-13T13:32:28.895449Z","shell.execute_reply":"2022-08-13T13:32:28.899208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = \"\"\"<div>\n<h1>Real or Fake</h1>\n<p>Kaggle </p>\n<a href=\"https://www.kaggle.com/c/nlp-getting-started\">getting started</a>\n</div>\"\"\"\n\nprint(remove_html(example))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.901710Z","iopub.execute_input":"2022-08-13T13:32:28.902100Z","iopub.status.idle":"2022-08-13T13:32:28.909987Z","shell.execute_reply.started":"2022-08-13T13:32:28.902067Z","shell.execute_reply":"2022-08-13T13:32:28.908975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text']=df['text'].apply(lambda x : remove_html(x))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.911121Z","iopub.execute_input":"2022-08-13T13:32:28.911637Z","iopub.status.idle":"2022-08-13T13:32:28.933723Z","shell.execute_reply.started":"2022-08-13T13:32:28.911606Z","shell.execute_reply":"2022-08-13T13:32:28.932868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Remove Emojis**","metadata":{}},{"cell_type":"code","source":"# Reference : https://gist.github.com/slowkow/7a7f61f495e3dbb7e3d767f97bd7304b\ndef remove_emoji(text):\n    emoji_pattern = re.compile(\"[\"\n                           u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                           u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                           u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                           u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                           u\"\\U00002702-\\U000027B0\"\n                           u\"\\U000024C2-\\U0001F251\"\n                           \"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r'', text)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.934986Z","iopub.execute_input":"2022-08-13T13:32:28.935879Z","iopub.status.idle":"2022-08-13T13:32:28.941650Z","shell.execute_reply.started":"2022-08-13T13:32:28.935844Z","shell.execute_reply":"2022-08-13T13:32:28.940596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text']=df['text'].apply(lambda x: remove_emoji(x))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.943199Z","iopub.execute_input":"2022-08-13T13:32:28.943833Z","iopub.status.idle":"2022-08-13T13:32:28.993836Z","shell.execute_reply.started":"2022-08-13T13:32:28.943799Z","shell.execute_reply":"2022-08-13T13:32:28.992988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Remove punctuations**","metadata":{}},{"cell_type":"code","source":"def remove_punct(text):\n    table=str.maketrans('','',string.punctuation)\n    return text.translate(table)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:28.995175Z","iopub.execute_input":"2022-08-13T13:32:28.995819Z","iopub.status.idle":"2022-08-13T13:32:29.000807Z","shell.execute_reply.started":"2022-08-13T13:32:28.995784Z","shell.execute_reply":"2022-08-13T13:32:28.999551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text']=df['text'].apply(lambda x : remove_punct(x))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:29.002410Z","iopub.execute_input":"2022-08-13T13:32:29.002807Z","iopub.status.idle":"2022-08-13T13:32:29.065732Z","shell.execute_reply.started":"2022-08-13T13:32:29.002749Z","shell.execute_reply":"2022-08-13T13:32:29.064930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Spelling Correction**","metadata":{}},{"cell_type":"code","source":"!pip install pyspellchecker","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-13T13:32:29.067443Z","iopub.execute_input":"2022-08-13T13:32:29.068080Z","iopub.status.idle":"2022-08-13T13:32:41.014326Z","shell.execute_reply.started":"2022-08-13T13:32:29.068042Z","shell.execute_reply":"2022-08-13T13:32:41.012989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from spellchecker import SpellChecker\n\nspell = SpellChecker()\ndef correct_spellings(text):\n    corrected_text = []\n    misspelled_words = spell.unknown(text.split())\n    for word in text.split():\n        if word in misspelled_words:\n            corrected_text.append(spell.correction(word))\n        else:\n            corrected_text.append(word)\n    return \" \".join(corrected_text)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:41.015853Z","iopub.execute_input":"2022-08-13T13:32:41.016263Z","iopub.status.idle":"2022-08-13T13:32:41.155878Z","shell.execute_reply.started":"2022-08-13T13:32:41.016213Z","shell.execute_reply":"2022-08-13T13:32:41.154886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Commenting out the spellchecker as it takes a lot of time to compute**","metadata":{}},{"cell_type":"code","source":"#df['text']=df['text'].apply(lambda x : correct_spellings(x))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:41.157483Z","iopub.execute_input":"2022-08-13T13:32:41.157885Z","iopub.status.idle":"2022-08-13T13:32:41.162853Z","shell.execute_reply.started":"2022-08-13T13:32:41.157845Z","shell.execute_reply":"2022-08-13T13:32:41.161838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **Vectorization with GloVe**","metadata":{}},{"cell_type":"code","source":"def create_corpus(df):\n    corpus=[]\n    for tweet in tqdm(df['text']):\n        words=[word.lower() for word in word_tokenize(tweet) if((word.isalpha()==1) & (word not in stop))]\n        corpus.append(words)\n    return corpus","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:41.164680Z","iopub.execute_input":"2022-08-13T13:32:41.165457Z","iopub.status.idle":"2022-08-13T13:32:41.173358Z","shell.execute_reply.started":"2022-08-13T13:32:41.165415Z","shell.execute_reply":"2022-08-13T13:32:41.172415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus=create_corpus(df)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:41.174633Z","iopub.execute_input":"2022-08-13T13:32:41.175153Z","iopub.status.idle":"2022-08-13T13:32:42.486052Z","shell.execute_reply.started":"2022-08-13T13:32:41.175119Z","shell.execute_reply":"2022-08-13T13:32:42.484990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_dict={}\nwith open('../input/glove-global-vectors-for-word-representation/glove.6B.100d.txt','r') as f:\n    for line in f:\n        values=line.split()\n        word=values[0]\n        vectors=np.asarray(values[1:],'float32')\n        embedding_dict[word]=vectors\nf.close()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:42.487747Z","iopub.execute_input":"2022-08-13T13:32:42.488142Z","iopub.status.idle":"2022-08-13T13:32:51.539377Z","shell.execute_reply.started":"2022-08-13T13:32:42.488106Z","shell.execute_reply":"2022-08-13T13:32:51.538361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_LEN=50\ntokenizer_obj=Tokenizer()\ntokenizer_obj.fit_on_texts(corpus)\nsequences=tokenizer_obj.texts_to_sequences(corpus)\n\ntweet_pad=pad_sequences(sequences,maxlen=MAX_LEN,truncating='post',padding='post')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:51.540718Z","iopub.execute_input":"2022-08-13T13:32:51.542252Z","iopub.status.idle":"2022-08-13T13:32:51.750617Z","shell.execute_reply.started":"2022-08-13T13:32:51.542213Z","shell.execute_reply":"2022-08-13T13:32:51.749636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_index=tokenizer_obj.word_index\nprint('Number of unique words:',len(word_index))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:51.755953Z","iopub.execute_input":"2022-08-13T13:32:51.756239Z","iopub.status.idle":"2022-08-13T13:32:51.762650Z","shell.execute_reply.started":"2022-08-13T13:32:51.756213Z","shell.execute_reply":"2022-08-13T13:32:51.761741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_words=len(word_index)+1\nembedding_matrix=np.zeros((num_words,100))\n\nfor word,i in tqdm(word_index.items()):\n    if i > num_words:\n        continue\n    \n    emb_vec=embedding_dict.get(word)\n    if emb_vec is not None:\n        embedding_matrix[i]=emb_vec","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:51.764286Z","iopub.execute_input":"2022-08-13T13:32:51.765004Z","iopub.status.idle":"2022-08-13T13:32:51.819577Z","shell.execute_reply.started":"2022-08-13T13:32:51.764966Z","shell.execute_reply":"2022-08-13T13:32:51.818593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **Baseline model**","metadata":{}},{"cell_type":"code","source":"model=Sequential()\n\nembedding=Embedding(num_words,100,embeddings_initializer=Constant(embedding_matrix),\n                   input_length=MAX_LEN,trainable=False)\n\nmodel.add(embedding)\nmodel.add(SpatialDropout1D(0.2))\nmodel.add(LSTM(64, dropout=0.2, recurrent_dropout=0.2))\nmodel.add(Dense(1, activation='sigmoid'))\n\n\noptimzer=Adam(learning_rate=1e-5)\n\nmodel.compile(loss='binary_crossentropy',optimizer=optimzer,metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:51.820853Z","iopub.execute_input":"2022-08-13T13:32:51.821783Z","iopub.status.idle":"2022-08-13T13:32:54.876079Z","shell.execute_reply.started":"2022-08-13T13:32:51.821741Z","shell.execute_reply":"2022-08-13T13:32:54.875142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:54.877595Z","iopub.execute_input":"2022-08-13T13:32:54.877961Z","iopub.status.idle":"2022-08-13T13:32:54.885418Z","shell.execute_reply.started":"2022-08-13T13:32:54.877926Z","shell.execute_reply":"2022-08-13T13:32:54.884366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=tweet_pad[:tweet.shape[0]]\ntest=tweet_pad[tweet.shape[0]:]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:54.886908Z","iopub.execute_input":"2022-08-13T13:32:54.888108Z","iopub.status.idle":"2022-08-13T13:32:54.897726Z","shell.execute_reply.started":"2022-08-13T13:32:54.888072Z","shell.execute_reply":"2022-08-13T13:32:54.896747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train,X_test,y_train,y_test=train_test_split(train,tweet['target'].values,test_size=0.15)\nprint('Shape of train',X_train.shape)\nprint(\"Shape of Validation \",X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:54.899618Z","iopub.execute_input":"2022-08-13T13:32:54.900014Z","iopub.status.idle":"2022-08-13T13:32:54.913147Z","shell.execute_reply.started":"2022-08-13T13:32:54.899981Z","shell.execute_reply":"2022-08-13T13:32:54.911953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history=model.fit(X_train,y_train,batch_size=4,epochs=15,validation_data=(X_test,y_test),verbose=2)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:32:54.914885Z","iopub.execute_input":"2022-08-13T13:32:54.915242Z","iopub.status.idle":"2022-08-13T14:51:19.214789Z","shell.execute_reply.started":"2022-08-13T13:32:54.915207Z","shell.execute_reply":"2022-08-13T14:51:19.213654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub=pd.read_csv('../input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T14:51:19.217046Z","iopub.execute_input":"2022-08-13T14:51:19.217488Z","iopub.status.idle":"2022-08-13T14:51:19.238356Z","shell.execute_reply.started":"2022-08-13T14:51:19.217446Z","shell.execute_reply":"2022-08-13T14:51:19.237430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pre=model.predict(test)\ny_pre=np.round(y_pre).astype(int).reshape(3263)\nsub=pd.DataFrame({'id':sample_sub['id'].values.tolist(),'target':y_pre})\nsub.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T14:51:19.239911Z","iopub.execute_input":"2022-08-13T14:51:19.240279Z","iopub.status.idle":"2022-08-13T14:51:20.676033Z","shell.execute_reply.started":"2022-08-13T14:51:19.240243Z","shell.execute_reply":"2022-08-13T14:51:20.674779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**If you liked this notebook, feel free to upvote, thank you!**","metadata":{}}]}