{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## !pip install pyspellchecker\n!pip install contractions","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:30:17.660105Z","iopub.execute_input":"2022-07-22T05:30:17.660538Z","iopub.status.idle":"2022-07-22T05:30:44.333248Z","shell.execute_reply.started":"2022-07-22T05:30:17.660437Z","shell.execute_reply":"2022-07-22T05:30:44.332149Z"}}},{"cell_type":"code","source":"!pip install pyspellchecker\n!pip install contractions","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport re\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nfrom nltk.corpus import stopwords\nfrom nltk.util import ngrams\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom collections import defaultdict\nfrom collections import  Counter\nplt.style.use('ggplot')\nstop=set(stopwords.words('english'))\nimport re\nfrom nltk.tokenize import word_tokenize\nimport gensim\nimport string\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom tqdm import tqdm\nfrom keras.models import Sequential\nfrom keras.layers import Embedding,LSTM,Dense,SpatialDropout1D\nfrom keras.initializers import Constant\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.optimizers import Adam\nfrom spellchecker import SpellChecker\nimport contractions\nfrom wordcloud import STOPWORDS\nfrom collections import defaultdict","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:37.617709Z","iopub.execute_input":"2022-07-28T02:23:37.61869Z","iopub.status.idle":"2022-07-28T02:23:49.631033Z","shell.execute_reply.started":"2022-07-28T02:23:37.618578Z","shell.execute_reply":"2022-07-28T02:23:49.626424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet= pd.read_csv('../input/nlp-getting-started/train.csv')\ntest=pd.read_csv('../input/nlp-getting-started/test.csv')\ntweet.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.632544Z","iopub.status.idle":"2022-07-28T02:23:49.632977Z","shell.execute_reply.started":"2022-07-28T02:23:49.632777Z","shell.execute_reply":"2022-07-28T02:23:49.632799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/nlp-getting-started/train.csv', dtype={'id': np.int16, 'target': np.int8})\ndf_test = pd.read_csv('../input/nlp-getting-started/test.csv', dtype={'id': np.int16})\n\n# 単語数\ndf_train['word_count'] = df_train['text'].apply(lambda x: len(str(x).split()))\ndf_test['word_count'] = df_test['text'].apply(lambda x: len(str(x).split()))\n\n# ユニークな単語数\ndf_train['unique_word_count'] = df_train['text'].apply(lambda x: len(set(str(x).split())))\ndf_test['unique_word_count'] = df_test['text'].apply(lambda x: len(set(str(x).split())))\n\n# ストップワードの数\ndf_train['stop_word_count'] = df_train['text'].apply(lambda x: len([w for w in str(x).lower().split() if w in STOPWORDS]))\ndf_test['stop_word_count'] = df_test['text'].apply(lambda x: len([w for w in str(x).lower().split() if w in STOPWORDS]))\n\n# URLの数\ndf_train['url_count'] = df_train['text'].apply(lambda x: len([w for w in str(x).lower().split() if 'http' in w or 'https' in w]))\ndf_test['url_count'] = df_test['text'].apply(lambda x: len([w for w in str(x).lower().split() if 'http' in w or 'https' in w]))\n\n# 単語文字数の平均\ndf_train['mean_word_length'] = df_train['text'].apply(lambda x: np.mean([len(w) for w in str(x).split()]))\ndf_test['mean_word_length'] = df_test['text'].apply(lambda x: np.mean([len(w) for w in str(x).split()]))\n\n# 文字数\ndf_train['char_count'] = df_train['text'].apply(lambda x: len(str(x)))\ndf_test['char_count'] = df_test['text'].apply(lambda x: len(str(x)))\n\n# 句読点の個数\ndf_train['punctuation_count'] = df_train['text'].apply(lambda x: len([c for c in str(x) if c in string.punctuation]))\ndf_test['punctuation_count'] = df_test['text'].apply(lambda x: len([c for c in str(x) if c in string.punctuation]))\n\n# ハッシュタグの個数\ndf_train['hashtag_count'] = df_train['text'].apply(lambda x: len([c for c in str(x) if c == '#']))\ndf_test['hashtag_count'] = df_test['text'].apply(lambda x: len([c for c in str(x) if c == '#']))\n\n# メンションの個数\ndf_train['mention_count'] = df_train['text'].apply(lambda x: len([c for c in str(x) if c == '@']))\ndf_test['mention_count'] = df_test['text'].apply(lambda x: len([c for c in str(x) if c == '@']))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.634625Z","iopub.status.idle":"2022-07-28T02:23:49.635048Z","shell.execute_reply.started":"2022-07-28T02:23:49.63484Z","shell.execute_reply":"2022-07-28T02:23:49.63486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 9つの特徴の分布を、災害ツイート=1 ⇄ 災害以外のツイート=0、訓練データ ⇄ テストデータで比較する\nMETAFEATURES = ['word_count', 'unique_word_count', 'stop_word_count', 'url_count', 'mean_word_length',\n                'char_count', 'punctuation_count', 'hashtag_count', 'mention_count']\nDISASTER_TWEETS = df_train['target'] == 1\n\nfig, axes = plt.subplots(ncols=2, nrows=len(METAFEATURES), figsize=(20, 50), dpi=100)\n\nfor i, feature in enumerate(METAFEATURES):\n    # 災害ツイート=1 ⇄ 災害以外のツイート=0の分布を比較する(カーネル密度推定を行う)\n    sns.distplot(df_train.loc[~DISASTER_TWEETS][feature], label='Not Disaster', ax=axes[i][0], color='green', kde=True)\n    sns.distplot(df_train.loc[DISASTER_TWEETS][feature], label='Disaster', ax=axes[i][0], color='red', kde=True)\n\n    # 訓練データ ⇄ テストデータの分布を比較する(カーネル密度推定を行う)\n    sns.distplot(df_train[feature], label='Training', ax=axes[i][1], kde=True)\n    sns.distplot(df_test[feature], label='Test', ax=axes[i][1], kde=True)\n\n    for j in range(2):\n        axes[i][j].set_xlabel('')\n        axes[i][j].tick_params(axis='x', labelsize=12)\n        axes[i][j].tick_params(axis='y', labelsize=12)\n        axes[i][j].legend()\n\n    axes[i][0].set_title(f'{feature} Target Distribution in Training Set', fontsize=13)\n    axes[i][1].set_title(f'{feature} Training & Test Set Distribution', fontsize=13)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.636516Z","iopub.status.idle":"2022-07-28T02:23:49.637073Z","shell.execute_reply.started":"2022-07-28T02:23:49.636875Z","shell.execute_reply":"2022-07-28T02:23:49.636896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# keywordの単語毎に、targetの平均値を求めて、その値を訓練データ全体に付加する\ndf_train['target_mean'] = df_train.groupby('keyword')['target'].transform('mean')\n\nfig = plt.figure(figsize=(8, 72), dpi=100)\n\n# keyword に含まれるラベル分布を確認\nsns.countplot(y=df_train.sort_values(by='target_mean', ascending=False)['keyword'],\n             hue=df_train.sort_values(by='target_mean', ascending=False)['target'])\n\nplt.tick_params(axis='x', labelsize=15)\nplt.tick_params(axis='y', labelsize=12)\nplt.legend(loc=1)\nplt.title('Target Distribution in Keywords')\n\nplt.show()\n\n# targetの値の平均値のカラムは以降使用しないので削除する\ndf_train.drop(columns=['target_mean'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.638731Z","iopub.status.idle":"2022-07-28T02:23:49.639141Z","shell.execute_reply.started":"2022-07-28T02:23:49.638929Z","shell.execute_reply":"2022-07-28T02:23:49.638948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.concat([tweet,test])\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.640955Z","iopub.status.idle":"2022-07-28T02:23:49.641592Z","shell.execute_reply.started":"2022-07-28T02:23:49.641377Z","shell.execute_reply":"2022-07-28T02:23:49.641404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_URL(text):\n    url = re.compile(r'https?://\\S+|www\\.\\S+')\n    return url.sub(r'',text)\n\ndef remove_html(text):\n    html=re.compile(r'<.*?>')\n    return html.sub(r'',text)\n\ndef remove_emoji(text):\n    emoji_pattern = re.compile(\"[\"\n                           u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                           u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                           u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                           u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                           u\"\\U00002702-\\U000027B0\"\n                           u\"\\U000024C2-\\U0001F251\"\n                           \"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r'', text)\n\ndef remove_punct(text):\n    table=str.maketrans('','',string.punctuation)\n    return text.translate(table)\n\ndef str_lower(text):\n    return text.lower()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.642797Z","iopub.status.idle":"2022-07-28T02:23:49.643233Z","shell.execute_reply.started":"2022-07-28T02:23:49.643013Z","shell.execute_reply":"2022-07-28T02:23:49.643033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text']=df['text'].apply(lambda x : remove_URL(x))\ndf['text']=df['text'].apply(lambda x : remove_html(x))\ndf['text']=df['text'].apply(lambda x: remove_emoji(x))\ndf['text']=df['text'].apply(lambda x : remove_punct(x))\ndf['text']=df['text'].apply(lambda x : str_lower(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.644777Z","iopub.status.idle":"2022-07-28T02:23:49.64523Z","shell.execute_reply.started":"2022-07-28T02:23:49.645009Z","shell.execute_reply":"2022-07-28T02:23:49.645029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spell = SpellChecker()\ndef correct_spellings(text):\n    corrected_text = []\n    misspelled_words = spell.unknown(text.split())\n    for word in text.split():\n        if word in misspelled_words:\n            corrected_text.append(spell.correction(word))\n        else:\n            corrected_text.append(word)\n    return \" \".join(corrected_text)\n        \ndef fix_contractions(text):\n    return contractions.fix(text)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.647045Z","iopub.status.idle":"2022-07-28T02:23:49.647679Z","shell.execute_reply.started":"2022-07-28T02:23:49.647471Z","shell.execute_reply":"2022-07-28T02:23:49.647492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text']=df['text'].apply(lambda x : correct_spellings(x))\ndf['text']=df['text'].apply(lambda x : fix_contractions(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.649086Z","iopub.status.idle":"2022-07-28T02:23:49.649513Z","shell.execute_reply.started":"2022-07-28T02:23:49.649313Z","shell.execute_reply":"2022-07-28T02:23:49.649332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_corpus(df):\n    corpus=[]\n    for tweet in tqdm(df['text']):\n        words=[word.lower() for word in word_tokenize(tweet) if((word.isalpha()==1) & (word not in stop))]\n        corpus.append(words)\n    return corpus\n\ncorpus=create_corpus(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.651244Z","iopub.status.idle":"2022-07-28T02:23:49.651729Z","shell.execute_reply.started":"2022-07-28T02:23:49.651531Z","shell.execute_reply":"2022-07-28T02:23:49.651551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_dict={}\nwith open('../input/glove-global-vectors-for-word-representation/glove.6B.200d.txt','r') as f:\n    for line in f:\n        values=line.split()\n        word=values[0]\n        vectors=np.asarray(values[1:],'float32')\n        embedding_dict[word]=vectors\nf.close()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.65299Z","iopub.status.idle":"2022-07-28T02:23:49.653399Z","shell.execute_reply.started":"2022-07-28T02:23:49.653204Z","shell.execute_reply":"2022-07-28T02:23:49.653225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_LEN=50\ntokenizer_obj=Tokenizer()\ntokenizer_obj.fit_on_texts(corpus)\nsequences=tokenizer_obj.texts_to_sequences(corpus)\n\ntweet_pad=pad_sequences(sequences,maxlen=MAX_LEN,truncating='post',padding='post')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.655156Z","iopub.status.idle":"2022-07-28T02:23:49.65556Z","shell.execute_reply.started":"2022-07-28T02:23:49.655371Z","shell.execute_reply":"2022-07-28T02:23:49.65539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_index=tokenizer_obj.word_index\nprint('Number of unique words:',len(word_index))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.657083Z","iopub.status.idle":"2022-07-28T02:23:49.657888Z","shell.execute_reply.started":"2022-07-28T02:23:49.657305Z","shell.execute_reply":"2022-07-28T02:23:49.657324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_words=len(word_index)+1\nembedding_matrix=np.zeros((num_words,200))\n\nfor word,i in tqdm(word_index.items()):\n    if i > num_words:\n        continue\n    \n    emb_vec=embedding_dict.get(word)\n    if emb_vec is not None:\n        embedding_matrix[i]=emb_vec","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.659411Z","iopub.status.idle":"2022-07-28T02:23:49.659764Z","shell.execute_reply.started":"2022-07-28T02:23:49.659589Z","shell.execute_reply":"2022-07-28T02:23:49.659606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=Sequential()\n\nembedding=Embedding(num_words,200,embeddings_initializer=Constant(embedding_matrix),\n                   input_length=MAX_LEN,trainable=False)\n\nmodel.add(embedding)\nmodel.add(SpatialDropout1D(0.2))\nmodel.add(LSTM(64, dropout=0.2, recurrent_dropout=0.2))\nmodel.add(Dense(1, activation='sigmoid'))\n\n\noptimzer=Adam(learning_rate=1e-5)\n\nmodel.compile(loss='binary_crossentropy',optimizer=optimzer,metrics=['accuracy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.661232Z","iopub.status.idle":"2022-07-28T02:23:49.661596Z","shell.execute_reply.started":"2022-07-28T02:23:49.661419Z","shell.execute_reply":"2022-07-28T02:23:49.661436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=tweet_pad[:tweet.shape[0]]\ntest=tweet_pad[tweet.shape[0]:]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.662704Z","iopub.status.idle":"2022-07-28T02:23:49.663052Z","shell.execute_reply.started":"2022-07-28T02:23:49.662877Z","shell.execute_reply":"2022-07-28T02:23:49.662894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train,X_test,y_train,y_test=train_test_split(train,tweet['target'].values,test_size=0.15)\nprint('Shape of train',X_train.shape)\nprint(\"Shape of Validation \",X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.664172Z","iopub.status.idle":"2022-07-28T02:23:49.664544Z","shell.execute_reply.started":"2022-07-28T02:23:49.664364Z","shell.execute_reply":"2022-07-28T02:23:49.664381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history=model.fit(X_train,y_train,batch_size=8,epochs=30,validation_data=(X_test,y_test),verbose=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.66574Z","iopub.status.idle":"2022-07-28T02:23:49.666153Z","shell.execute_reply.started":"2022-07-28T02:23:49.665942Z","shell.execute_reply":"2022-07-28T02:23:49.665961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub=pd.read_csv('../input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.667291Z","iopub.status.idle":"2022-07-28T02:23:49.667667Z","shell.execute_reply.started":"2022-07-28T02:23:49.667481Z","shell.execute_reply":"2022-07-28T02:23:49.667498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pre=model.predict(test)\nprint(y_pre)\ny_pre=np.round(y_pre).astype(int).reshape(3263)\nprint(y_pre)\nsub=pd.DataFrame({'id':sample_sub['id'].values.tolist(),'target':y_pre})\nsub.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.668727Z","iopub.status.idle":"2022-07-28T02:23:49.669106Z","shell.execute_reply.started":"2022-07-28T02:23:49.668922Z","shell.execute_reply":"2022-07-28T02:23:49.66894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:23:49.670659Z","iopub.status.idle":"2022-07-28T02:23:49.671052Z","shell.execute_reply.started":"2022-07-28T02:23:49.670864Z","shell.execute_reply":"2022-07-28T02:23:49.670883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}