{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pyspellchecker\n!pip install contractions","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:30:17.660105Z","iopub.execute_input":"2022-07-22T05:30:17.660538Z","iopub.status.idle":"2022-07-22T05:30:44.333248Z","shell.execute_reply.started":"2022-07-22T05:30:17.660437Z","shell.execute_reply":"2022-07-22T05:30:44.332149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport re\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nfrom nltk.corpus import stopwords\nfrom nltk.util import ngrams\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom collections import defaultdict\nfrom collections import  Counter\nplt.style.use('ggplot')\nstop=set(stopwords.words('english'))\nimport re\nfrom nltk.tokenize import word_tokenize\nimport gensim\nimport string\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom tqdm import tqdm\nfrom keras.models import Sequential\nfrom keras.layers import Embedding,LSTM,Dense,SpatialDropout1D\nfrom keras.initializers import Constant\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.optimizers import Adam\nfrom spellchecker import SpellChecker\nimport contractions\nfrom wordcloud import STOPWORDS\nfrom collections import defaultdict","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:30:44.335541Z","iopub.execute_input":"2022-07-22T05:30:44.335889Z","iopub.status.idle":"2022-07-22T05:30:56.250070Z","shell.execute_reply.started":"2022-07-22T05:30:44.335855Z","shell.execute_reply":"2022-07-22T05:30:56.249038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet= pd.read_csv('../input/nlp-getting-started/train.csv')\ntest=pd.read_csv('../input/nlp-getting-started/test.csv')\ntweet.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:30:56.252118Z","iopub.execute_input":"2022-07-22T05:30:56.253272Z","iopub.status.idle":"2022-07-22T05:30:56.347635Z","shell.execute_reply.started":"2022-07-22T05:30:56.253223Z","shell.execute_reply":"2022-07-22T05:30:56.346637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/nlp-getting-started/train.csv', dtype={'id': np.int16, 'target': np.int8})\ndf_test = pd.read_csv('../input/nlp-getting-started/test.csv', dtype={'id': np.int16})\n\n# 単語数\ndf_train['word_count'] = df_train['text'].apply(lambda x: len(str(x).split()))\ndf_test['word_count'] = df_test['text'].apply(lambda x: len(str(x).split()))\n\n# ユニークな単語数\ndf_train['unique_word_count'] = df_train['text'].apply(lambda x: len(set(str(x).split())))\ndf_test['unique_word_count'] = df_test['text'].apply(lambda x: len(set(str(x).split())))\n\n# ストップワードの数\ndf_train['stop_word_count'] = df_train['text'].apply(lambda x: len([w for w in str(x).lower().split() if w in STOPWORDS]))\ndf_test['stop_word_count'] = df_test['text'].apply(lambda x: len([w for w in str(x).lower().split() if w in STOPWORDS]))\n\n# URLの数\ndf_train['url_count'] = df_train['text'].apply(lambda x: len([w for w in str(x).lower().split() if 'http' in w or 'https' in w]))\ndf_test['url_count'] = df_test['text'].apply(lambda x: len([w for w in str(x).lower().split() if 'http' in w or 'https' in w]))\n\n# 単語文字数の平均\ndf_train['mean_word_length'] = df_train['text'].apply(lambda x: np.mean([len(w) for w in str(x).split()]))\ndf_test['mean_word_length'] = df_test['text'].apply(lambda x: np.mean([len(w) for w in str(x).split()]))\n\n# 文字数\ndf_train['char_count'] = df_train['text'].apply(lambda x: len(str(x)))\ndf_test['char_count'] = df_test['text'].apply(lambda x: len(str(x)))\n\n# 句読点の個数\ndf_train['punctuation_count'] = df_train['text'].apply(lambda x: len([c for c in str(x) if c in string.punctuation]))\ndf_test['punctuation_count'] = df_test['text'].apply(lambda x: len([c for c in str(x) if c in string.punctuation]))\n\n# ハッシュタグの個数\ndf_train['hashtag_count'] = df_train['text'].apply(lambda x: len([c for c in str(x) if c == '#']))\ndf_test['hashtag_count'] = df_test['text'].apply(lambda x: len([c for c in str(x) if c == '#']))\n\n# メンションの個数\ndf_train['mention_count'] = df_train['text'].apply(lambda x: len([c for c in str(x) if c == '@']))\ndf_test['mention_count'] = df_test['text'].apply(lambda x: len([c for c in str(x) if c == '@']))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:30:56.349855Z","iopub.execute_input":"2022-07-22T05:30:56.350383Z","iopub.status.idle":"2022-07-22T05:30:57.038660Z","shell.execute_reply.started":"2022-07-22T05:30:56.350349Z","shell.execute_reply":"2022-07-22T05:30:57.037249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 9つの特徴の分布を、災害ツイート=1 ⇄ 災害以外のツイート=0、訓練データ ⇄ テストデータで比較する\nMETAFEATURES = ['word_count', 'unique_word_count', 'stop_word_count', 'url_count', 'mean_word_length',\n                'char_count', 'punctuation_count', 'hashtag_count', 'mention_count']\nDISASTER_TWEETS = df_train['target'] == 1\n\nfig, axes = plt.subplots(ncols=2, nrows=len(METAFEATURES), figsize=(20, 50), dpi=100)\n\nfor i, feature in enumerate(METAFEATURES):\n    # 災害ツイート=1 ⇄ 災害以外のツイート=0の分布を比較する(カーネル密度推定を行う)\n    sns.distplot(df_train.loc[~DISASTER_TWEETS][feature], label='Not Disaster', ax=axes[i][0], color='green', kde=True)\n    sns.distplot(df_train.loc[DISASTER_TWEETS][feature], label='Disaster', ax=axes[i][0], color='red', kde=True)\n\n    # 訓練データ ⇄ テストデータの分布を比較する(カーネル密度推定を行う)\n    sns.distplot(df_train[feature], label='Training', ax=axes[i][1], kde=True)\n    sns.distplot(df_test[feature], label='Test', ax=axes[i][1], kde=True)\n\n    for j in range(2):\n        axes[i][j].set_xlabel('')\n        axes[i][j].tick_params(axis='x', labelsize=12)\n        axes[i][j].tick_params(axis='y', labelsize=12)\n        axes[i][j].legend()\n\n    axes[i][0].set_title(f'{feature} Target Distribution in Training Set', fontsize=13)\n    axes[i][1].set_title(f'{feature} Training & Test Set Distribution', fontsize=13)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:30:57.040395Z","iopub.execute_input":"2022-07-22T05:30:57.041109Z","iopub.status.idle":"2022-07-22T05:31:04.697019Z","shell.execute_reply.started":"2022-07-22T05:30:57.041060Z","shell.execute_reply":"2022-07-22T05:31:04.695673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# keywordの単語毎に、targetの平均値を求めて、その値を訓練データ全体に付加する\ndf_train['target_mean'] = df_train.groupby('keyword')['target'].transform('mean')\n\nfig = plt.figure(figsize=(8, 72), dpi=100)\n\n# keyword に含まれるラベル分布を確認\nsns.countplot(y=df_train.sort_values(by='target_mean', ascending=False)['keyword'],\n             hue=df_train.sort_values(by='target_mean', ascending=False)['target'])\n\nplt.tick_params(axis='x', labelsize=15)\nplt.tick_params(axis='y', labelsize=12)\nplt.legend(loc=1)\nplt.title('Target Distribution in Keywords')\n\nplt.show()\n\n# targetの値の平均値のカラムは以降使用しないので削除する\ndf_train.drop(columns=['target_mean'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:04.698568Z","iopub.execute_input":"2022-07-22T05:31:04.699230Z","iopub.status.idle":"2022-07-22T05:31:08.748736Z","shell.execute_reply.started":"2022-07-22T05:31:04.699164Z","shell.execute_reply":"2022-07-22T05:31:08.747169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.concat([tweet,test])\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:08.750487Z","iopub.execute_input":"2022-07-22T05:31:08.751690Z","iopub.status.idle":"2022-07-22T05:31:08.764732Z","shell.execute_reply.started":"2022-07-22T05:31:08.751639Z","shell.execute_reply":"2022-07-22T05:31:08.763464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_URL(text):\n    url = re.compile(r'https?://\\S+|www\\.\\S+')\n    return url.sub(r'',text)\n\ndef remove_html(text):\n    html=re.compile(r'<.*?>')\n    return html.sub(r'',text)\n\ndef remove_emoji(text):\n    emoji_pattern = re.compile(\"[\"\n                           u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                           u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                           u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                           u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                           u\"\\U00002702-\\U000027B0\"\n                           u\"\\U000024C2-\\U0001F251\"\n                           \"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r'', text)\n\ndef remove_punct(text):\n    table=str.maketrans('','',string.punctuation)\n    return text.translate(table)\n\ndef str_lower(text):\n    return text.lower()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:08.767227Z","iopub.execute_input":"2022-07-22T05:31:08.768246Z","iopub.status.idle":"2022-07-22T05:31:08.779302Z","shell.execute_reply.started":"2022-07-22T05:31:08.768194Z","shell.execute_reply":"2022-07-22T05:31:08.778133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text']=df['text'].apply(lambda x : remove_URL(x))\ndf['text']=df['text'].apply(lambda x : remove_html(x))\ndf['text']=df['text'].apply(lambda x: remove_emoji(x))\ndf['text']=df['text'].apply(lambda x : remove_punct(x))\ndf['text']=df['text'].apply(lambda x : str_lower(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:08.782598Z","iopub.execute_input":"2022-07-22T05:31:08.784512Z","iopub.status.idle":"2022-07-22T05:31:08.791123Z","shell.execute_reply.started":"2022-07-22T05:31:08.784455Z","shell.execute_reply":"2022-07-22T05:31:08.790250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spell = SpellChecker()\ndef correct_spellings(text):\n    corrected_text = []\n    misspelled_words = spell.unknown(text.split())\n    for word in text.split():\n        if word in misspelled_words:\n            corrected_text.append(spell.correction(word))\n        else:\n            corrected_text.append(word)\n    return \" \".join(corrected_text)\n        \ndef fix_contractions(text):\n    return contractions.fix(text)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:08.797986Z","iopub.execute_input":"2022-07-22T05:31:08.798855Z","iopub.status.idle":"2022-07-22T05:31:08.959410Z","shell.execute_reply.started":"2022-07-22T05:31:08.798815Z","shell.execute_reply":"2022-07-22T05:31:08.958189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text']=df['text'].apply(lambda x : correct_spellings(x))\ndf['text']=df['text'].apply(lambda x : fix_contractions(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:08.960885Z","iopub.execute_input":"2022-07-22T05:31:08.961402Z","iopub.status.idle":"2022-07-22T05:31:08.966828Z","shell.execute_reply.started":"2022-07-22T05:31:08.961356Z","shell.execute_reply":"2022-07-22T05:31:08.965506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_corpus(df):\n    corpus=[]\n    for tweet in tqdm(df['text']):\n        words=[word.lower() for word in word_tokenize(tweet) if((word.isalpha()==1) & (word not in stop))]\n        corpus.append(words)\n    return corpus\n\ncorpus=create_corpus(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:08.968628Z","iopub.execute_input":"2022-07-22T05:31:08.969230Z","iopub.status.idle":"2022-07-22T05:31:12.681822Z","shell.execute_reply.started":"2022-07-22T05:31:08.969168Z","shell.execute_reply":"2022-07-22T05:31:12.680538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_dict={}\nwith open('../input/glove-global-vectors-for-word-representation/glove.6B.200d.txt','r') as f:\n    for line in f:\n        values=line.split()\n        word=values[0]\n        vectors=np.asarray(values[1:],'float32')\n        embedding_dict[word]=vectors\nf.close()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:12.683364Z","iopub.execute_input":"2022-07-22T05:31:12.683741Z","iopub.status.idle":"2022-07-22T05:31:39.159796Z","shell.execute_reply.started":"2022-07-22T05:31:12.683705Z","shell.execute_reply":"2022-07-22T05:31:39.158632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_LEN=50\ntokenizer_obj=Tokenizer()\ntokenizer_obj.fit_on_texts(corpus)\nsequences=tokenizer_obj.texts_to_sequences(corpus)\n\ntweet_pad=pad_sequences(sequences,maxlen=MAX_LEN,truncating='post',padding='post')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:39.161473Z","iopub.execute_input":"2022-07-22T05:31:39.161925Z","iopub.status.idle":"2022-07-22T05:31:39.443238Z","shell.execute_reply.started":"2022-07-22T05:31:39.161880Z","shell.execute_reply":"2022-07-22T05:31:39.442022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_index=tokenizer_obj.word_index\nprint('Number of unique words:',len(word_index))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:39.444626Z","iopub.execute_input":"2022-07-22T05:31:39.445492Z","iopub.status.idle":"2022-07-22T05:31:39.451066Z","shell.execute_reply.started":"2022-07-22T05:31:39.445450Z","shell.execute_reply":"2022-07-22T05:31:39.450246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_words=len(word_index)+1\nembedding_matrix=np.zeros((num_words,200))\n\nfor word,i in tqdm(word_index.items()):\n    if i > num_words:\n        continue\n    \n    emb_vec=embedding_dict.get(word)\n    if emb_vec is not None:\n        embedding_matrix[i]=emb_vec","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:39.452599Z","iopub.execute_input":"2022-07-22T05:31:39.453195Z","iopub.status.idle":"2022-07-22T05:31:39.574280Z","shell.execute_reply.started":"2022-07-22T05:31:39.453136Z","shell.execute_reply":"2022-07-22T05:31:39.573334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=Sequential()\n\nembedding=Embedding(num_words,200,embeddings_initializer=Constant(embedding_matrix),\n                   input_length=MAX_LEN,trainable=False)\n\nmodel.add(embedding)\nmodel.add(SpatialDropout1D(0.2))\nmodel.add(LSTM(64, dropout=0.2, recurrent_dropout=0.2))\nmodel.add(Dense(1, activation='sigmoid'))\n\n\noptimzer=Adam(learning_rate=1e-5)\n\nmodel.compile(loss='binary_crossentropy',optimizer=optimzer,metrics=['accuracy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:39.575561Z","iopub.execute_input":"2022-07-22T05:31:39.576588Z","iopub.status.idle":"2022-07-22T05:31:39.989895Z","shell.execute_reply.started":"2022-07-22T05:31:39.576551Z","shell.execute_reply":"2022-07-22T05:31:39.988562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=tweet_pad[:tweet.shape[0]]\ntest=tweet_pad[tweet.shape[0]:]","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:39.991892Z","iopub.execute_input":"2022-07-22T05:31:39.992770Z","iopub.status.idle":"2022-07-22T05:31:39.999258Z","shell.execute_reply.started":"2022-07-22T05:31:39.992721Z","shell.execute_reply":"2022-07-22T05:31:39.998231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train,X_test,y_train,y_test=train_test_split(train,tweet['target'].values,test_size=0.15)\nprint('Shape of train',X_train.shape)\nprint(\"Shape of Validation \",X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:40.000750Z","iopub.execute_input":"2022-07-22T05:31:40.001206Z","iopub.status.idle":"2022-07-22T05:31:40.013729Z","shell.execute_reply.started":"2022-07-22T05:31:40.001160Z","shell.execute_reply":"2022-07-22T05:31:40.012554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history=model.fit(X_train,y_train,batch_size=8,epochs=30,validation_data=(X_test,y_test),verbose=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:31:40.015199Z","iopub.execute_input":"2022-07-22T05:31:40.015561Z","iopub.status.idle":"2022-07-22T06:03:16.502542Z","shell.execute_reply.started":"2022-07-22T05:31:40.015530Z","shell.execute_reply":"2022-07-22T06:03:16.501356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub=pd.read_csv('../input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T06:03:16.504485Z","iopub.execute_input":"2022-07-22T06:03:16.505706Z","iopub.status.idle":"2022-07-22T06:03:16.536471Z","shell.execute_reply.started":"2022-07-22T06:03:16.505655Z","shell.execute_reply":"2022-07-22T06:03:16.535445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pre=model.predict(test)\nprint(y_pre)\ny_pre=np.round(y_pre).astype(int).reshape(3263)\nprint(y_pre)\nsub=pd.DataFrame({'id':sample_sub['id'].values.tolist(),'target':y_pre})\nsub.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T06:03:20.565278Z","iopub.execute_input":"2022-07-22T06:03:20.566066Z","iopub.status.idle":"2022-07-22T06:03:24.390473Z","shell.execute_reply.started":"2022-07-22T06:03:20.566018Z","shell.execute_reply":"2022-07-22T06:03:24.389228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T06:03:24.392221Z","iopub.execute_input":"2022-07-22T06:03:24.392639Z","iopub.status.idle":"2022-07-22T06:03:24.404395Z","shell.execute_reply.started":"2022-07-22T06:03:24.392603Z","shell.execute_reply":"2022-07-22T06:03:24.402955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}