{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 必要なライブラリをインポート","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nfrom nltk.corpus import stopwords\nfrom nltk.util import ngrams\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom collections import defaultdict\nfrom collections import  Counter\nplt.style.use('ggplot')\nstop=set(stopwords.words('english'))\nimport re\nfrom nltk.tokenize import word_tokenize\nimport gensim\nimport string\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom tqdm import tqdm\nfrom keras.models import Sequential\nfrom keras.layers import Embedding,LSTM,Dense,SpatialDropout1D\nfrom keras.initializers import Constant\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:35.450789Z","iopub.execute_input":"2022-07-15T07:18:35.451327Z","iopub.status.idle":"2022-07-15T07:18:41.280931Z","shell.execute_reply.started":"2022-07-15T07:18:35.451211Z","shell.execute_reply":"2022-07-15T07:18:41.279994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 訓練データと検証データを読み込む\nツイートの一部を表示して確認してみる","metadata":{}},{"cell_type":"code","source":"tweet = pd.read_csv('../input/nlp-getting-started/train.csv')\ntest = pd.read_csv('../input/nlp-getting-started/test.csv')\ntweet","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:41.282949Z","iopub.execute_input":"2022-07-15T07:18:41.283632Z","iopub.status.idle":"2022-07-15T07:18:41.366851Z","shell.execute_reply.started":"2022-07-15T07:18:41.283597Z","shell.execute_reply":"2022-07-15T07:18:41.365954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:41.368813Z","iopub.execute_input":"2022-07-15T07:18:41.369531Z","iopub.status.idle":"2022-07-15T07:18:41.384268Z","shell.execute_reply.started":"2022-07-15T07:18:41.369493Z","shell.execute_reply":"2022-07-15T07:18:41.383127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = tweet.target.value_counts()\nsns.barplot(x.index, x)\nplt.gca().set_ylabel('samples')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:41.387649Z","iopub.execute_input":"2022-07-15T07:18:41.388184Z","iopub.status.idle":"2022-07-15T07:18:41.563160Z","shell.execute_reply.started":"2022-07-15T07:18:41.388159Z","shell.execute_reply":"2022-07-15T07:18:41.561988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,(ax1, ax2) = plt.subplots(1, 2, figsize = (10, 5))\ntweet_len = tweet[tweet['target'] == 1]['text'].str.len()\nax1.hist(tweet_len, color = 'red')\nax1.set_title('disaster tweets')\ntweet_len = tweet[tweet['target'] == 0]['text'].str.len()\nax2.hist(tweet_len, color = 'green')\nax2.set_title('Not disaster tweets')\nfig.suptitle('Characters in tweets')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:41.568241Z","iopub.execute_input":"2022-07-15T07:18:41.568621Z","iopub.status.idle":"2022-07-15T07:18:41.883097Z","shell.execute_reply.started":"2022-07-15T07:18:41.568584Z","shell.execute_reply":"2022-07-15T07:18:41.882085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\ntweet_len=tweet[tweet['target']==1]['text'].str.split().map(lambda x: len(x))\nax1.hist(tweet_len,color='red')\nax1.set_title('disaster tweets')\ntweet_len=tweet[tweet['target']==0]['text'].str.split().map(lambda x: len(x))\nax2.hist(tweet_len,color='green')\nax2.set_title('Not disaster tweets')\nfig.suptitle('Words in a tweet')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:41.884391Z","iopub.execute_input":"2022-07-15T07:18:41.885111Z","iopub.status.idle":"2022-07-15T07:18:42.224740Z","shell.execute_reply.started":"2022-07-15T07:18:41.885051Z","shell.execute_reply":"2022-07-15T07:18:42.223903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\nword=tweet[tweet['target']==1]['text'].str.split().apply(lambda x : [len(i) for i in x])\nsns.distplot(word.map(lambda x: np.mean(x)),ax=ax1,color='red')\nax1.set_title('disaster')\nword=tweet[tweet['target']==0]['text'].str.split().apply(lambda x : [len(i) for i in x])\nsns.distplot(word.map(lambda x: np.mean(x)),ax=ax2,color='green')\nax2.set_title('Not disaster')\nfig.suptitle('Average word length in each tweet')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:42.226263Z","iopub.execute_input":"2022-07-15T07:18:42.226610Z","iopub.status.idle":"2022-07-15T07:18:42.892170Z","shell.execute_reply.started":"2022-07-15T07:18:42.226575Z","shell.execute_reply":"2022-07-15T07:18:42.891130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ツイートの分を単語ごとに分ける関数\ndef create_corpus(target):\n    corpus = []\n    \n    for x in tweet[tweet['target'] == target]['text'].str.split() :\n        for i in x :\n            corpus.append(i)\n    return corpus","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:42.893484Z","iopub.execute_input":"2022-07-15T07:18:42.894306Z","iopub.status.idle":"2022-07-15T07:18:42.901131Z","shell.execute_reply.started":"2022-07-15T07:18:42.894262Z","shell.execute_reply":"2022-07-15T07:18:42.899928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus=create_corpus(1)\n\ndic=defaultdict(int)\n\nprint(dic.keys())","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:42.902904Z","iopub.execute_input":"2022-07-15T07:18:42.903296Z","iopub.status.idle":"2022-07-15T07:18:42.925017Z","shell.execute_reply.started":"2022-07-15T07:18:42.903238Z","shell.execute_reply":"2022-07-15T07:18:42.924057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus=create_corpus(1)\n\ndic=defaultdict(int)\nfor word in corpus:\n    if word in stop:\n        dic[word]+=1\n\ntop=sorted(dic.items(), key=lambda x:x[1],reverse=True)[:10] \n\nx,y=zip(*top)\nplt.bar(x,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:42.929272Z","iopub.execute_input":"2022-07-15T07:18:42.929514Z","iopub.status.idle":"2022-07-15T07:18:43.155790Z","shell.execute_reply.started":"2022-07-15T07:18:42.929491Z","shell.execute_reply":"2022-07-15T07:18:43.154705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:43.157207Z","iopub.execute_input":"2022-07-15T07:18:43.157594Z","iopub.status.idle":"2022-07-15T07:18:43.165195Z","shell.execute_reply.started":"2022-07-15T07:18:43.157553Z","shell.execute_reply":"2022-07-15T07:18:43.164037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:43.167376Z","iopub.execute_input":"2022-07-15T07:18:43.168335Z","iopub.status.idle":"2022-07-15T07:18:43.175861Z","shell.execute_reply.started":"2022-07-15T07:18:43.168297Z","shell.execute_reply":"2022-07-15T07:18:43.174642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\ncorpus = create_corpus(1)\n\ndic = defaultdict(int)\nimport string\nspecial = string.punctuation\nfor i in (corpus):\n    if i in special:\n        dic[i] += 1\n        \nx,y = zip(*dic.items())\nplt.bar(x,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:43.178982Z","iopub.execute_input":"2022-07-15T07:18:43.180619Z","iopub.status.idle":"2022-07-15T07:18:43.538773Z","shell.execute_reply.started":"2022-07-15T07:18:43.180565Z","shell.execute_reply":"2022-07-15T07:18:43.537764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10, 5))\ncorpus = create_corpus(0)\n\ndic = defaultdict(int)\nimport string \nspecial = string.punctuation\nfor i in (corpus):\n    if i in special:\n        dic[i] += 1\n        \nx, y = zip(*dic.items())\nplt.bar(x, y, color = 'green')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:43.543373Z","iopub.execute_input":"2022-07-15T07:18:43.545725Z","iopub.status.idle":"2022-07-15T07:18:43.915858Z","shell.execute_reply.started":"2022-07-15T07:18:43.545685Z","shell.execute_reply":"2022-07-15T07:18:43.914766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter = Counter(corpus)\nmost = counter.most_common()\nx = []\ny = []\n\nfor word, count in most[:40]:\n    if(word not in stop):\n        x.append(word)\n        y.append(count)\n        \nsns.barplot(x = y, y = x)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:43.921240Z","iopub.execute_input":"2022-07-15T07:18:43.923733Z","iopub.status.idle":"2022-07-15T07:18:44.217321Z","shell.execute_reply.started":"2022-07-15T07:18:43.923693Z","shell.execute_reply":"2022-07-15T07:18:44.216386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_top_tweet_bigrams(corpus, n = None):\n    vec = CountVectorizer(ngram_range = (2, 2)).fit(corpus)\n    bag_of_words = vec.transform(corpus)\n    sum_words = bag_of_words.sum(axis = 0)\n    words_freq=[(word, sum_words[0, idx]) for word, idx in vec.vocabulary_.items()]\n    words_freq = sorted(words_freq, key = lambda x : x[1], reverse = True)\n    return words_freq[:n]\n\n\nplt.figure(figsize = (10, 5))\ntop_tweet_bigrams = get_top_tweet_bigrams(tweet['text'])[:10]\nx, y = map(list, zip(*top_tweet_bigrams))\nsns.barplot(x = y, y = x)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:44.221531Z","iopub.execute_input":"2022-07-15T07:18:44.224082Z","iopub.status.idle":"2022-07-15T07:18:45.797970Z","shell.execute_reply.started":"2022-07-15T07:18:44.224014Z","shell.execute_reply":"2022-07-15T07:18:45.796953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.concat([tweet, test])\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:45.802728Z","iopub.execute_input":"2022-07-15T07:18:45.805375Z","iopub.status.idle":"2022-07-15T07:18:45.820036Z","shell.execute_reply.started":"2022-07-15T07:18:45.805329Z","shell.execute_reply":"2022-07-15T07:18:45.818963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = \"New competition launched :https://www.kaggle.com/c/nlp-getting-started\"\n\ndef remove_URL(text):\n    url = re.compile(r'https?://\\S+|www\\.\\S+')\n    return url.sub(r'', text)\n\ndf['text'] = df['text'].apply(lambda x :remove_URL(x))\n\nprint(remove_URL(example))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:45.824763Z","iopub.execute_input":"2022-07-15T07:18:45.827203Z","iopub.status.idle":"2022-07-15T07:18:45.903614Z","shell.execute_reply.started":"2022-07-15T07:18:45.827165Z","shell.execute_reply":"2022-07-15T07:18:45.902557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = \"\"\"<div>\n<h1>Real or Fake</h1>\n<p>Kaggle </p>\n<a href=\"https://www.kaggle.com/c/nlp-getting-started\">getting started</a>\n</div>\"\"\"\n\ndef remove_html(text):\n    html = re.compile(r'<.*?>')\n    return html.sub(r'',text)\nprint(remove_html(example))\n\ndf['text'] = df['text'].apply(lambda x : remove_html(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:45.907756Z","iopub.execute_input":"2022-07-15T07:18:45.910214Z","iopub.status.idle":"2022-07-15T07:18:45.965282Z","shell.execute_reply.started":"2022-07-15T07:18:45.910171Z","shell.execute_reply":"2022-07-15T07:18:45.964338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reference : https://gist.github.com/slowkow/7a7f61f495e3dbb7e3d767f97bd7304b\ndef remove_emoji(text):\n    emoji_pattern = re.compile(\"[\"\n                           u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                           u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                           u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                           u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                           u\"\\U00002702-\\U000027B0\"\n                           u\"\\U000024C2-\\U0001F251\"\n                           \"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r'', text)\n\ndf['text'] = df['text'].apply(lambda x: remove_emoji(x))\n\nprint(remove_emoji(\"Omg another Earthquake 😔😔\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:45.969519Z","iopub.execute_input":"2022-07-15T07:18:45.972145Z","iopub.status.idle":"2022-07-15T07:18:46.093611Z","shell.execute_reply.started":"2022-07-15T07:18:45.972107Z","shell.execute_reply":"2022-07-15T07:18:46.092753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_punct(text):\n    table = str.maketrans('','', string.punctuation)\n    return text.translate(table)\n\nexample = \"I am a #king\"\nprint(remove_punct(example))\n\ndf['text'] = df['text'].apply(lambda x : remove_punct(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:46.097605Z","iopub.execute_input":"2022-07-15T07:18:46.099890Z","iopub.status.idle":"2022-07-15T07:18:46.190432Z","shell.execute_reply.started":"2022-07-15T07:18:46.099854Z","shell.execute_reply":"2022-07-15T07:18:46.189478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyspellchecker","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:18:46.194739Z","iopub.execute_input":"2022-07-15T07:18:46.196911Z","iopub.status.idle":"2022-07-15T07:19:01.728949Z","shell.execute_reply.started":"2022-07-15T07:18:46.196874Z","shell.execute_reply":"2022-07-15T07:19:01.727922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from spellchecker import SpellChecker\n\nspell = SpellChecker()\ndef correct_spellings(text):\n    corrected_text = []\n    misspelled_words = spell.unknown(text.split())\n    for word in text.split():\n        if word in misspelled_words:\n            corrected_text.append(spell.correction(word))\n        else:\n            corrected_text.append(word)\n    return \" \".join(corrected_text)\n        \ntext = \"corect me plese\"\ncorrect_spellings(text)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:19:01.732604Z","iopub.execute_input":"2022-07-15T07:19:01.732895Z","iopub.status.idle":"2022-07-15T07:19:01.873966Z","shell.execute_reply.started":"2022-07-15T07:19:01.732867Z","shell.execute_reply":"2022-07-15T07:19:01.872975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_corpus(df):\n    corpus = []\n    for tweet in tqdm(df['text']):\n        words = [word.lower() for word in word_tokenize(tweet) if((word.isalpha() == 1) & (word not in stop))]\n        corpus.append(words)\n    return corpus\n\ncorpus = create_corpus(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:19:01.875576Z","iopub.execute_input":"2022-07-15T07:19:01.876123Z","iopub.status.idle":"2022-07-15T07:19:04.332446Z","shell.execute_reply.started":"2022-07-15T07:19:01.876085Z","shell.execute_reply":"2022-07-15T07:19:04.331132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_dict = {}\nwith open('../input/glove6b/glove.6B.300d.txt','r') as f:\n    for line in f:\n        values = line.split()\n        word = values[0]\n        vectors = np.asarray(values[1:],'float32')\n        embedding_dict[word]  =vectors\nf.close()\n\nMAX_LEN = 50\ntokenizer_obj = Tokenizer()\ntokenizer_obj.fit_on_texts(corpus)\nsequences = tokenizer_obj.texts_to_sequences(corpus)\n\ntweet_pad = pad_sequences(sequences, maxlen = MAX_LEN, truncating = 'post', padding = 'post')\n\nword_index = tokenizer_obj.word_index\nprint('Number of unique words:', len(word_index))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:19:04.336768Z","iopub.execute_input":"2022-07-15T07:19:04.337264Z","iopub.status.idle":"2022-07-15T07:19:39.693228Z","shell.execute_reply.started":"2022-07-15T07:19:04.337225Z","shell.execute_reply":"2022-07-15T07:19:39.692245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_words = len(word_index) + 1\nembedding_matrix = np.zeros((num_words, 300))\n\nfor word,i in tqdm(word_index.items()):\n    if i > num_words:\n        continue\n    \n    emb_vec = embedding_dict.get(word)\n    if emb_vec is not None:\n        embedding_matrix[i] = emb_vec","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:19:39.694737Z","iopub.execute_input":"2022-07-15T07:19:39.695093Z","iopub.status.idle":"2022-07-15T07:19:39.776255Z","shell.execute_reply.started":"2022-07-15T07:19:39.695040Z","shell.execute_reply":"2022-07-15T07:19:39.775311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\n\nembedding=Embedding(num_words, 300, embeddings_initializer=Constant(embedding_matrix),\n                   input_length=MAX_LEN, trainable=False)\n\nmodel.add(embedding)\nmodel.add(SpatialDropout1D(0.2))\nmodel.add(LSTM(64, dropout=0.2, recurrent_dropout=0.2))\nmodel.add(Dense(1, activation='sigmoid'))\n\noptimzer = Adam(learning_rate=1e-5)\n\nmodel.compile(loss='binary_crossentropy',optimizer=optimzer,metrics=['accuracy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:19:39.777877Z","iopub.execute_input":"2022-07-15T07:19:39.778540Z","iopub.status.idle":"2022-07-15T07:19:42.523280Z","shell.execute_reply.started":"2022-07-15T07:19:39.778500Z","shell.execute_reply":"2022-07-15T07:19:42.522288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=tweet_pad[:tweet.shape[0]]\ntest=tweet_pad[tweet.shape[0]:]\n\nX_train,X_test,y_train,y_test = train_test_split(train,tweet['target'].values,test_size=0.15)\nprint('Shape of train',X_train.shape)\nprint(\"Shape of Validation \",X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:19:42.528549Z","iopub.execute_input":"2022-07-15T07:19:42.528822Z","iopub.status.idle":"2022-07-15T07:19:42.541097Z","shell.execute_reply.started":"2022-07-15T07:19:42.528796Z","shell.execute_reply":"2022-07-15T07:19:42.540119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history=model.fit(X_train,y_train,batch_size=4,epochs=15,validation_data=(X_test,y_test),verbose=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:19:42.542795Z","iopub.execute_input":"2022-07-15T07:19:42.543041Z","iopub.status.idle":"2022-07-15T08:41:06.860758Z","shell.execute_reply.started":"2022-07-15T07:19:42.543017Z","shell.execute_reply":"2022-07-15T08:41:06.859756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub=pd.read_csv('../input/nlp-getting-started/sample_submission.csv')\n\ny_pre = model.predict(test)\ny_pre = np.round(y_pre).astype(int).reshape(3263)\nsub = pd.DataFrame({'id':sample_sub['id'].values.tolist(), 'target':y_pre})\nsub.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T09:16:43.203628Z","iopub.execute_input":"2022-07-15T09:16:43.204063Z","iopub.status.idle":"2022-07-15T09:16:45.164515Z","shell.execute_reply.started":"2022-07-15T09:16:43.204025Z","shell.execute_reply":"2022-07-15T09:16:45.163155Z"},"trusted":true},"execution_count":null,"outputs":[]}]}